diff --git a/benchmarks/patches/vllm/minimax_m3_nvfp4_marlin.patch b/benchmarks/patches/vllm/minimax_m3_nvfp4_marlin.patch new file mode 100644 index 0000000000..a44bf655da --- /dev/null +++ b/benchmarks/patches/vllm/minimax_m3_nvfp4_marlin.patch @@ -0,0 +1,30 @@ +--- a/vllm/model_executor/layers/fused_moe/oracle/nvfp4.py ++++ b/vllm/model_executor/layers/fused_moe/oracle/nvfp4.py +@@ -176,2 +176,3 @@ + NvFp4MoeBackend.FLASHINFER_TRTLLM, ++ NvFp4MoeBackend.MARLIN, + } +--- a/vllm/model_executor/layers/fused_moe/experts/marlin_moe.py ++++ b/vllm/model_executor/layers/fused_moe/experts/marlin_moe.py +@@ -583,9 +583,12 @@ +- # Gated-activation params (used by SWIGLUOAI_UNINTERLEAVE on packed w13). +- # silu == swigluoai with alpha=1, beta=0; configs that don't set these +- # (plain silu) fall back to the silu identity. +- self.gemm1_alpha = ( +- quant_config.gemm1_alpha if quant_config.gemm1_alpha is not None else 1.0 +- ) +- self.gemm1_beta = ( +- quant_config.gemm1_beta if quant_config.gemm1_beta is not None else 0.0 +- ) ++ if self.gemm1_clamp_limit is None: ++ self.gemm1_clamp_limit = moe_config.swiglu_limit ++ ++ gemm1_alpha = quant_config.gemm1_alpha ++ if gemm1_alpha is None: ++ gemm1_alpha = moe_config.swiglu_alpha ++ self.gemm1_alpha = 1.0 if gemm1_alpha is None else gemm1_alpha ++ ++ gemm1_beta = quant_config.gemm1_beta ++ if gemm1_beta is None: ++ gemm1_beta = moe_config.swiglu_beta ++ self.gemm1_beta = 0.0 if gemm1_beta is None else gemm1_beta diff --git a/benchmarks/single_node/fixed_seq_len/minimaxm3_fp4_rtx6000pro.sh b/benchmarks/single_node/fixed_seq_len/minimaxm3_fp4_rtx6000pro.sh new file mode 100755 index 0000000000..8f2aa5dc6b --- /dev/null +++ b/benchmarks/single_node/fixed_seq_len/minimaxm3_fp4_rtx6000pro.sh @@ -0,0 +1,112 @@ +#!/usr/bin/env bash + +# MiniMax-M3 NVFP4 RTX PRO 6000 Blackwell single-node vLLM recipe. +# This is the PCIe/SM120 counterpart to minimaxm3_fp4_b200.sh. It keeps +# the ModelOpt NVFP4, FP8 KV-cache, and MSA block-size settings while using +# NCCL collectives instead of the B200-tuned FlashInfer/TRT-LLM all-reduce. +# +# The pinned vLLM image does not contain the MiniMax-M3 Marlin fixes tracked +# by vLLM PRs #45836 and #48929. Apply the narrow compatibility patch covered +# by docs/waiver/2306.md before importing vLLM. + +source "$(dirname "$0")/../../benchmark_lib.sh" + +check_env_vars \ + MODEL \ + TP \ + EP_SIZE \ + DP_ATTENTION \ + CONC \ + ISL \ + OSL \ + MAX_MODEL_LEN \ + RANDOM_RANGE_RATIO \ + RESULT_FILENAME + +if [[ "$MODEL" != /* ]]; then hf download "$MODEL"; fi + +if [[ -n "$SLURM_JOB_ID" ]]; then + echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" +fi + +nvidia-smi + +SERVER_LOG=/workspace/server.log +GPU_MEM_UTIL="${GPU_MEM_UTIL:-0.90}" + +export VLLM_ENGINE_READY_TIMEOUT_S=3600 +export VLLM_FLOAT32_MATMUL_PRECISION=high + +if [ "${DP_ATTENTION}" = "true" ]; then + PARALLEL_ARGS=( + --tensor-parallel-size 1 + --data-parallel-size "$TP" + --enable-expert-parallel + ) +elif [ "$EP_SIZE" -gt 1 ]; then + PARALLEL_ARGS=( + --tensor-parallel-size "$TP" + --enable-expert-parallel + ) +else + PARALLEL_ARGS=(--tensor-parallel-size "$TP") +fi + +if [ "${EVAL_ONLY}" = "true" ]; then + setup_eval_context + MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" +fi +start_gpu_monitor + +VLLM_SITE_PACKAGES="$( + python3 -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])' +)" +VLLM_MARLIN_PATCH=/workspace/benchmarks/patches/vllm/minimax_m3_nvfp4_marlin.patch +if ! patch --batch --forward --fuzz=0 -p1 -d "$VLLM_SITE_PACKAGES" \ + < "$VLLM_MARLIN_PATCH"; then + echo "Failed to apply the pinned MiniMax-M3 Marlin compatibility patch" >&2 + exit 1 +fi + +set -x +vllm serve "$MODEL" --port "$PORT" \ + "${PARALLEL_ARGS[@]}" \ + --disable-custom-all-reduce \ + --gpu-memory-utilization "$GPU_MEM_UTIL" \ + --max-model-len "$MAX_MODEL_LEN" \ + --kv-cache-dtype fp8 \ + --block-size 128 \ + --language-model-only \ + --attention-backend TRITON_ATTN \ + --moe-backend marlin \ + --max-cudagraph-capture-size 2048 \ + --max-num-seqs "$CONC" \ + --max-num-batched-tokens "$((ISL * 2))" \ + --stream-interval 20 \ + --no-enable-prefix-caching \ + --trust-remote-code > "$SERVER_LOG" 2>&1 & + +SERVER_PID=$! + +wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" + +run_benchmark_serving \ + --model "$MODEL" \ + --port "$PORT" \ + --backend vllm \ + --input-len "$ISL" \ + --output-len "$OSL" \ + --random-range-ratio "$RANDOM_RANGE_RATIO" \ + --num-prompts "$((CONC * 10))" \ + --max-concurrency "$CONC" \ + --result-filename "$RESULT_FILENAME" \ + --result-dir /workspace/ \ + --trust-remote-code + +if [ "${RUN_EVAL}" = "true" ]; then + run_eval --framework lm-eval --port "$PORT" + append_lm_eval_summary +fi + +stop_gpu_monitor +set +x diff --git a/benchmarks/single_node/fixed_seq_len/minimaxm3_fp4_rtx6000pro_mtp.sh b/benchmarks/single_node/fixed_seq_len/minimaxm3_fp4_rtx6000pro_mtp.sh new file mode 100755 index 0000000000..618578a58b --- /dev/null +++ b/benchmarks/single_node/fixed_seq_len/minimaxm3_fp4_rtx6000pro_mtp.sh @@ -0,0 +1,140 @@ +#!/usr/bin/env bash + +# MiniMax-M3 NVFP4 RTX PRO 6000 Blackwell single-node vLLM recipe with +# EAGLE3 speculative decoding. The SM120 target and draft both use Triton +# attention; the target's NVFP4 experts use Marlin. +# +# The pinned vLLM image does not contain the MiniMax-M3 Marlin fixes tracked +# by vLLM PRs #45836 and #48929. Apply the narrow compatibility patch covered +# by docs/waiver/2306.md before importing vLLM. + +source "$(dirname "$0")/../../benchmark_lib.sh" + +check_env_vars \ + MODEL \ + TP \ + EP_SIZE \ + DP_ATTENTION \ + CONC \ + ISL \ + OSL \ + MAX_MODEL_LEN \ + RANDOM_RANGE_RATIO \ + RESULT_FILENAME + +SERVED_MODEL_NAME="$MODEL" +TARGET_MODEL_PATH="${MODEL_PATH:-$MODEL}" +if [[ "$TARGET_MODEL_PATH" != /* ]]; then + hf download "$TARGET_MODEL_PATH" +fi + +DRAFT_MODEL="Inferact/MiniMax-M3-EAGLE3" +hf download "$DRAFT_MODEL" + +if [[ -n "$SLURM_JOB_ID" ]]; then + echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME" +fi + +nvidia-smi + +SERVER_LOG=/workspace/server.log +GPU_MEM_UTIL="${GPU_MEM_UTIL:-0.90}" +NUM_SPEC_TOKENS="${NUM_SPEC_TOKENS:-3}" + +if [[ ! "$NUM_SPEC_TOKENS" =~ ^[1-9][0-9]*$ ]]; then + echo "NUM_SPEC_TOKENS must be a positive integer, got: $NUM_SPEC_TOKENS" >&2 + exit 1 +fi + +# vLLM's speculative decode capture size is measured in tokens. Each active +# sequence can contribute one accepted token plus NUM_SPEC_TOKENS draft tokens. +CUDAGRAPH_CAPTURE_SIZE="$((CONC * (NUM_SPEC_TOKENS + 1)))" + +export VLLM_ENGINE_READY_TIMEOUT_S=3600 +export VLLM_FLOAT32_MATMUL_PRECISION=high + +if [ "${DP_ATTENTION}" = "true" ]; then + PARALLEL_ARGS=( + --tensor-parallel-size 1 + --data-parallel-size "$TP" + --enable-expert-parallel + ) + DRAFT_TP=1 +elif [ "$EP_SIZE" -gt 1 ]; then + PARALLEL_ARGS=( + --tensor-parallel-size "$TP" + --enable-expert-parallel + ) + DRAFT_TP="$TP" +else + PARALLEL_ARGS=(--tensor-parallel-size "$TP") + DRAFT_TP="$TP" +fi + +if [ "${EVAL_ONLY}" = "true" ]; then + setup_eval_context + MAX_MODEL_LEN="$EVAL_MAX_MODEL_LEN" +fi +start_gpu_monitor + +VLLM_SITE_PACKAGES="$( + python3 -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])' +)" +VLLM_MARLIN_PATCH=/workspace/benchmarks/patches/vllm/minimax_m3_nvfp4_marlin.patch +if ! patch --batch --forward --fuzz=0 -p1 -d "$VLLM_SITE_PACKAGES" \ + < "$VLLM_MARLIN_PATCH"; then + echo "Failed to apply the pinned MiniMax-M3 Marlin compatibility patch" >&2 + exit 1 +fi + +SPECULATIVE_CONFIG="$( + printf '{"method":"eagle3","model":"%s","num_speculative_tokens":%d,"draft_tensor_parallel_size":%d,"attention_backend":"TRITON_ATTN"}' \ + "$DRAFT_MODEL" "$NUM_SPEC_TOKENS" "$DRAFT_TP" +)" + +set -x +vllm serve "$TARGET_MODEL_PATH" \ + --served-model-name "$SERVED_MODEL_NAME" \ + --port "$PORT" \ + "${PARALLEL_ARGS[@]}" \ + --disable-custom-all-reduce \ + --gpu-memory-utilization "$GPU_MEM_UTIL" \ + --max-model-len "$MAX_MODEL_LEN" \ + --kv-cache-dtype fp8 \ + --block-size 128 \ + --language-model-only \ + --attention-backend TRITON_ATTN \ + --moe-backend marlin \ + --max-cudagraph-capture-size "$CUDAGRAPH_CAPTURE_SIZE" \ + --max-num-seqs "$CONC" \ + --max-num-batched-tokens "$((ISL * 2))" \ + --speculative-config "$SPECULATIVE_CONFIG" \ + --stream-interval 20 \ + --no-enable-prefix-caching \ + --trust-remote-code > "$SERVER_LOG" 2>&1 & + +SERVER_PID=$! + +wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID" + +run_benchmark_serving \ + --model "$SERVED_MODEL_NAME" \ + --port "$PORT" \ + --backend vllm \ + --input-len "$ISL" \ + --output-len "$OSL" \ + --random-range-ratio "$RANDOM_RANGE_RATIO" \ + --num-prompts "$((CONC * 10))" \ + --max-concurrency "$CONC" \ + --result-filename "$RESULT_FILENAME" \ + --result-dir /workspace/ \ + --trust-remote-code \ + --use-chat-template + +if [ "${RUN_EVAL}" = "true" ]; then + run_eval --framework lm-eval --port "$PORT" + append_lm_eval_summary +fi + +stop_gpu_monitor +set +x diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index a388cad95d..cd0455d58f 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -7347,6 +7347,44 @@ minimaxm3-fp8-b300-vllm: - { tp: 4, ep: 4, dp-attn: true, conc-start: 64, conc-end: 128 } - { tp: 8, ep: 8, dp-attn: true, conc-start: 128, conc-end: 512 } +# MiniMax-M3 NVFP4 single-node vLLM sweep using 4 of 8 RTX PRO 6000 GPUs. +minimaxm3-fp4-rtx6000pro-vllm: + image: vllm/vllm-openai:vllm-minimax-m3-perf-x86_64-13.0.1-8b00f41@sha256:6af4be7ae69a5f424de85b3d514ac79778bdcf4a9d05f93ec908a4095e0f1253 + model: nvidia/MiniMax-M3-NVFP4 + model-prefix: minimaxm3 + runner: rtx6000pro-lat + precision: fp4 + framework: vllm + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - { tp: 4, conc-list: [1, 4, 16, 64] } + # Marlin EP4 uses the standard NCCL expert exchange; DeepEP/NIXL + # batched activation layouts are not compatible. + - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64] } + +# EAGLE3 speculative-decoding twin of the RTX PRO 6000 MiniMax-M3 NVFP4 +# sweep. The external Inferact/MiniMax-M3-EAGLE3 draft uses three speculative +# tokens and Triton attention on SM120; prompts use the model's chat template. +minimaxm3-fp4-rtx6000pro-vllm-mtp: + image: vllm/vllm-openai:vllm-minimax-m3-perf-x86_64-13.0.1-8b00f41@sha256:6af4be7ae69a5f424de85b3d514ac79778bdcf4a9d05f93ec908a4095e0f1253 + model: nvidia/MiniMax-M3-NVFP4 + model-prefix: minimaxm3 + runner: rtx6000pro-lat + precision: fp4 + framework: vllm + multinode: false + scenarios: + fixed-seq-len: + - isl: 8192 + osl: 1024 + search-space: + - { tp: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } + - { tp: 4, ep: 4, conc-list: [1, 4, 16, 64], spec-decoding: mtp } + # MiniMax-M3 NVFP4 (nvidia/MiniMax-M3-NVFP4) B300 single-node vLLM — FP4 variant # of minimaxm3-fp8-b300-vllm. MiniMax-M3 modelopt NVFP4 support (vllm-project/vllm # PR #46380) is baked into the perf container image, so no runtime patch is diff --git a/configs/runners.yaml b/configs/runners.yaml index 851b821ba2..4c6b11069d 100644 --- a/configs/runners.yaml +++ b/configs/runners.yaml @@ -164,6 +164,10 @@ labels: - gb300-nv_0 - gb300-nv_1 - gb300-nv_2 + rtx6000pro: + - rtx6000pro-lat_00 + rtx6000pro-lat: + - rtx6000pro-lat_00 cluster:h100-cw: - h100-cw_00 - h100-cw_01 @@ -251,6 +255,8 @@ labels: - gb300-nv_0 - gb300-nv_1 - gb300-nv_2 + cluster:rtx6000pro-lat: + - rtx6000pro-lat_00 cluster:mi300x-amds: - mi300x-amds_00 - mi300x-amds_01 @@ -315,6 +321,9 @@ hardware: cluster:gb200-nv: available-cpu-dram-mib: 860_160 gpus-per-node: 4 + cluster:rtx6000pro-lat: + available-cpu-dram-mib: 1_500_000 + gpus-per-node: 8 cluster:mi300x-amds: available-cpu-dram-mib: 2_321_924 gpus-per-node: 8 diff --git a/docs/waiver/2306.md b/docs/waiver/2306.md new file mode 100644 index 0000000000..4703704d42 --- /dev/null +++ b/docs/waiver/2306.md @@ -0,0 +1,47 @@ +# Runtime patch waiver for PR 2306 + +