diff --git a/MODELS.md b/MODELS.md index 3fc822420e..15de10cea6 100644 --- a/MODELS.md +++ b/MODELS.md @@ -18,7 +18,7 @@ This document tracks every model benchmarked by InferenceX-e2e: when it was adde | Model architecture class | Prefix | Date added | Active scenarios | Deprecated scenarios | |---|---|---|---|---| | Qwen3.8 2.4T | `qwen3.8` | TBD | Agentic coding | | -| Kimi-K3 | `kimik3` | 2026-07-27 | Agentic coding | | +| Kimi-K3 | `kimik3` | 2026-07-27 ([#2370](https://github.com/SemiAnalysisAI/InferenceX/pull/2370)) | Agentic coding | | | GLM-5.2 | `glm5.2` | 2026-07-18 ([#2268](https://github.com/SemiAnalysisAI/InferenceX/pull/2268)) | Agentic coding | | | MiniMax-M3 | `minimaxm3` | 2026-06-12 ([#1724](https://github.com/SemiAnalysisAI/InferenceX/pull/1724)) | Single-turn 8k1k, Agentic coding | Single-turn 1k1k | | DeepSeek-V4-Pro | `dsv4` | 2026-04-24 ([#1130](https://github.com/SemiAnalysisAI/InferenceX/pull/1130)) | Single-turn 8k1k, Agentic coding | Single-turn 1k1k | diff --git a/MODELS_zh.md b/MODELS_zh.md index 177a4b4ba3..9381475e14 100644 --- a/MODELS_zh.md +++ b/MODELS_zh.md @@ -18,7 +18,7 @@ | 模型架构类别 | 前缀 | 加入日期 | 启用场景 | 已弃用场景 | |---|---|---|---|---| | Qwen3.8 2.4T | `qwen3.8` | 待定 | 智能体编码 | | -| Kimi-K3 | `kimik3` | 2026-07-27 | 智能体编码 | | +| Kimi-K3 | `kimik3` | 2026-07-27 ([#2370](https://github.com/SemiAnalysisAI/InferenceX/pull/2370)) | 智能体编码 | | | GLM-5.2 | `glm5.2` | 2026-07-18([#2268](https://github.com/SemiAnalysisAI/InferenceX/pull/2268)) | 智能体编码 | | | MiniMax-M3 | `minimaxm3` | 2026-06-12([#1724](https://github.com/SemiAnalysisAI/InferenceX/pull/1724)) | 单轮 8k1k、智能体编码 | 单轮 1k1k | | DeepSeek-V4-Pro | `dsv4` | 2026-04-24([#1130](https://github.com/SemiAnalysisAI/InferenceX/pull/1130)) | 单轮 8k1k、智能体编码 | 单轮 1k1k | diff --git a/benchmarks/multi_node/srt-slurm-recipes/configs/kimi-k3-container-deps.sh b/benchmarks/multi_node/srt-slurm-recipes/configs/kimi-k3-container-deps.sh new file mode 100644 index 0000000000..abddf4d9c3 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/configs/kimi-k3-container-deps.sh @@ -0,0 +1,39 @@ +#!/bin/bash +# Setup script for the Kimi-K3 vLLM bring-up image (vllm/vllm-openai:kimi-k3). +# srt-slurm runs this in every worker container before dynamo install and +# worker startup (recipe field: setup_script). + +set -euo pipefail + +# The image's first decode step crashes in the KDA hybrid-state postprocess: +# vllm/v1/worker/gpu/model_states/mamba_hybrid.py, postprocess_state: +# IndexError: index_fill_(): Expected dtype int64 for index. +# torch's index_fill_ requires an int64 index tensor, but the runner passes +# the int32 idx_mapping (hit by moonshotai/Kimi-K3 agentic bring-up, first +# decode step, engine v0.1.dev19262+gb6bbf29dd). Coerce the index to int64. +# Idempotent: exits 0 if the patch is already applied. +python3 - <<'PY' +import pathlib +import re + +import vllm.v1.worker.gpu.model_states.mamba_hybrid as mh + +path = pathlib.Path(mh.__file__) +src = path.read_text() +if "idx_mapping.long()" in src: + print(f"mamba_hybrid index_fill_ patch already applied: {path}") + raise SystemExit(0) + +new, n = re.subn( + r"index_fill_\(\s*0,\s*idx_mapping,", + "index_fill_(0, idx_mapping.long(),", + src, +) +if n != 1: + raise SystemExit( + f"expected exactly one index_fill_(0, idx_mapping, ...) call in " + f"{path}, found {n} — image layout changed, refusing to patch" + ) +path.write_text(new) +print(f"Patched mamba_hybrid index_fill_ index dtype: {path}") +PY diff --git a/benchmarks/multi_node/srt-slurm-recipes/patches/srt-slurm-pr278-direct-vllm-multinode.patch b/benchmarks/multi_node/srt-slurm-recipes/patches/srt-slurm-pr278-direct-vllm-multinode.patch new file mode 100644 index 0000000000..9408bc7764 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/patches/srt-slurm-pr278-direct-vllm-multinode.patch @@ -0,0 +1,62 @@ +diff --git a/src/srtctl/backends/vllm.py b/src/srtctl/backends/vllm.py +index 74f673b..377606a 100644 +--- a/src/srtctl/backends/vllm.py ++++ b/src/srtctl/backends/vllm.py +@@ -716,25 +716,31 @@ class VLLMProtocol: + if frontend_type == "vllm": + if mode != "agg": + raise ValueError("frontend.type: vllm supports aggregate vLLM jobs only") +- if is_multi_node: +- raise ValueError("frontend.type: vllm currently supports single-node aggregate jobs only") + + config.pop("host", None) + config.pop("port", None) + config.pop("connector", None) + config.setdefault("served-model-name", served_model_name) + +- cmd.extend( +- [ +- "vllm", +- "serve", +- model_arg, +- "--host", +- "0.0.0.0", +- "--port", +- str(runtime.frontend_port), +- ] +- ) ++ node_rank = endpoint_nodes.index(process.node) ++ cmd.extend(["vllm", "serve", model_arg]) ++ if node_rank == 0: ++ cmd.extend(["--host", "0.0.0.0", "--port", str(runtime.frontend_port)]) ++ if is_multi_node: ++ # vLLM-native multi-node serve (torchrun-style): the leader owns ++ # the OpenAI server; other node ranks run headless engine workers. ++ cmd.extend( ++ [ ++ "--master-addr", ++ leader_ip, ++ "--nnodes", ++ str(len(endpoint_nodes)), ++ "--node-rank", ++ str(node_rank), ++ ] ++ ) ++ if node_rank > 0: ++ cmd.append("--headless") + if not self.set_cuda_visible_devices: + device_ids = ",".join(str(i) for i in sorted(process.gpu_indices)) + if device_ids: +diff --git a/src/srtctl/core/schema.py b/src/srtctl/core/schema.py +index 1263ddc..0ef7ae4 100644 +--- a/src/srtctl/core/schema.py ++++ b/src/srtctl/core/schema.py +@@ -1587,8 +1587,6 @@ class SrtConfig: + raise ValidationError("frontend.type: vllm supports aggregate jobs only, not disaggregated layouts") + if self.resources.num_agg < 1: + raise ValidationError("frontend.type: vllm requires resources.agg_workers >= 1") +- if (self.resources.agg_nodes or 1) != 1: +- raise ValidationError("frontend.type: vllm currently supports single-node aggregate jobs only") + + def _validate_het_jobs(self): + """When ``resources.het_jobs`` is set to True, enforce supported shape. diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-agentic.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-agentic.yaml new file mode 100644 index 0000000000..0ba9cfc1a1 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-agentic.yaml @@ -0,0 +1,165 @@ +name: "kimik3-vllm-agg-b200-tp8pp2-simplecpu-agentic" + +# Kimi-K3 MXFP4 B200 AGGREGATED TP8 x PP2 agentic recipe (2 nodes / 16 GPUs). +# The native MXFP4 checkpoint (2.8T total params, ~1.4TB of weights) does not +# fit one 8xB200 node, so TP8 shards attention/dense (/8) and PP2 splits the +# 93 layers (/2) across 16 GPUs. Plain TP (NOT TEP): expert parallelism is +# deliberately off, so the 896 routed experts are TP-sharded inside each +# pipeline stage. Node allocation = tp*pp/gpus_per_node = 8*2/8 = 2 nodes. +# Aggregated (single worker, decode num-worker 0) — no P/D split, no NIXL. +# VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION fuses the K3 LatentMoE tail path in +# the kimi-k3 bring-up image. +model: + path: "kimik3" + container: "vllm/vllm-openai:kimi-k3" + precision: "fp4" + +identity: + model: + repo: "moonshotai/Kimi-K3" + container: + image: "vllm/vllm-openai:kimi-k3" + +# Direct vLLM serving (frontend.type: vllm, srt-slurm PR #278 + the +# InferenceX multinode patch): `vllm serve` owns the OpenAI port itself, so +# no Dynamo frontend/worker is involved and no dynamo install is needed. +dynamo: + install: false + +# Patches the image's mamba_hybrid postprocess_state: torch index_fill_ +# requires an int64 index but the runner passes the int32 idx_mapping, +# crashing the first decode step (IndexError: Expected dtype int64 for index). +setup_script: kimi-k3-container-deps.sh + +environment: + # SimpleCPUOffloadConnector: identical prefixes must hash to identical + # block keys across ranks. + PYTHONHASHSEED: "42" + +slurm: + time_limit: "8:00:00" + +health_check: + interval_seconds: 10 + max_attempts: 1440 + +resources: + gpu_type: "b200" + gpus_per_node: 8 + agg_nodes: 2 + agg_workers: 1 + gpus_per_agg: 16 + +infra: + etcd_nats_dedicated_node: false + nats_max_payload_mb: 32 + +frontend: + # Direct vLLM OpenAI server (srt-slurm PR #278): the vllm serve leader owns + # the public port; rank-1 runs a headless engine worker (vLLM-native + # multi-node TP8xPP2 via --master-addr/--nnodes/--node-rank, enabled by + # patches/srt-slurm-pr278-direct-vllm-multinode.patch). + type: vllm + enable_multiple_frontends: false + +backend: + type: vllm + connector: null + aggregated_environment: + VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" + VLLM_SERVER_DEV_MODE: "1" + # ~1.4TB of MXFP4 weights off shared Lustre: keep the engine-ready window + # generous, and let one long AgentX request hold a PP stage beyond vLLM's + # 300-second model-execution default. + VLLM_ENGINE_READY_TIMEOUT_S: "3600" + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" + # Evidence from THIS variant (conc16, SimpleCPU offload): a 2.36 GiB MLA + # prefill transient OOM'd while 2.63 GiB sat reserved-but-unallocated — + # allocator fragmentation, exactly what this mode fixes (and what the + # torch OOM message recommends). The GPU-resident variant D runs green + # without it; the offload connector's staging shifts the layout enough + # to fragment. NCCL_CUMEM_ENABLE stays unchanged. + PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" + # No VLLM_PREFIX_CACHE_RETENTION_INTERVAL: the GB200/GB300 AgentX value + # (32768) hard-fails engine init on Kimi-K3 — the KDA hybrid gives it a + # scheduler_block_size of 3145728 and the interval must be a multiple of + # it ("VLLM_PREFIX_CACHE_RETENTION_INTERVAL (32768) must be non-negative + # and a multiple of scheduler_block_size (3145728)"). Default retention + # served fine in earlier runs. + NCCL_CUMEM_ENABLE: "1" + TILELANG_CLEANUP_TEMP_FILES: "1" + UCX_MEMTYPE_CACHE: "n" + UCX_MEMTYPE_REG_WHOLE: "n" + UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1,mlx5_4:1,mlx5_5:1,mlx5_10:1,mlx5_11:1" + HF_HUB_CACHE: "/hf_hub_cache" + HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" + vllm_config: + aggregated: + served-model-name: "moonshotai/Kimi-K3" + tensor-parallel-size: 8 + pipeline-parallel-size: 2 + trust-remote-code: true + # SimpleCPUOffloadConnector (variant E): offload KV blocks to CPU DRAM. + # enable_cross_layers_blocks false: with true, KV-cache init crashes on + # K3's KDA hybrid — "shape '[64980, 64, 576]' is invalid for input of + # size 199618560", a 12x mismatch equal to the MLA layers per PP stage + # (24/2); the cross-layer block folding gets the hybrid geometry wrong. + # The GB200 recipes also run connectors with cross-layer blocks off. + # cpu_bytes_to_use_per_rank = TOTAL_CPU_DRAM_GB * 1e9 / GPU_COUNT with + # the framework's own budget math for cluster:b200-dgxc — DRAM + # min(3,095,781 MiB, 2,861,022 MiB cap) * 0.80 utilization = 2399 GB + # per node (matrix total-cpu-dram-gb), / 8 ranks = 299,875,000,000 + # bytes per rank. + kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use_per_rank":299875000000,"lazy_offload":false,"enable_cross_layers_blocks":false}}' + # Explicit (default-on) — offloaded blocks are only useful if prefix + # caching reuses them across agentic turns. + enable-prefix-caching: true + load-format: fastsafetensors + moe-backend: auto + # 0.90, not 0.95: the flashinfer trtllm MXFP4 MoE kernel allocates a + # ~1.6 GiB runtime workspace OUTSIDE vLLM's memory pool on the first + # forward; at 0.95 a 178 GiB B200 has only ~1.35 GiB free and the first + # warmup request OOMs (seen on the dynamo-frontend variants). 0.90 + # matches the GB200/GB300 agentic recipes. + gpu-memory-utilization: 0.90 + no-enable-flashinfer-autotune: true + # kimi_k3 parsers via the native vllm serve OpenAI-frontend flags — + # legitimate here because this recipe serves directly with vllm serve + # (frontend.type: vllm), not through the dynamo worker entrypoint that + # rejects them. + enable-auto-tool-choice: true + tool-call-parser: kimi_k3 + reasoning-parser: kimi_k3 + # No explicit max-model-len: let vLLM derive the native 1M window from + # the model config (agentic trajectories blow past any small cap, and + # K3's KDA layers keep per-token KV small — only the 24 gated-MLA + # layers hold cache). Prefix caching stays on (default) for trajectory + # reuse. Cap prefill chunks so a single long request cannot OOM a + # pipeline stage; let vLLM pick max-num-seqs. + max-num-batched-tokens: 8192 + +sbatch_directives: + segment: "1" + +srun_options: + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" + RESULT_DIR: "/logs/agentic" + PORT: "8000" + # Keep the aggregate worker in the multinode result schema so ingestion + # uses the zero decode-worker count instead of duplicating TP into P and D. + IS_MULTINODE: "true" + # aiperf's conv-aware routing emits nvext.session_control, a removed POC + # field this dynamo build 400-rejects at warmup (schema moved to + # router/routing_constraints/agent_hints). Same opt-out as the GB300 + # aggregate AgentX recipes — and with a single aggregate worker there is + # no P/D routing to bind anyway. + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" + WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 36ae816809..6a1d9de30b 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8325,3 +8325,54 @@ qwen3.5-fp8-gb200-dynamo-sglang-mtp: tp: 16 ep: 16 dp-attn: true + +# Kimi-K3 MXFP4 B200 aggregated vLLM via Dynamo (TP8 x PP2, 2 nodes / 16 +# GPUs), agentic bring-up. The native MXFP4 checkpoint (2.8T total params, +# ~1.4TB weights) does not fit one 8xB200 node, so TP8 shards attention/dense +# and PP2 splits layers. Plain TP (NOT TEP): ep 1, no expert parallelism — +# the 896 routed experts are TP-sharded within each pipeline stage. Node +# count = tp*pp/gpus_per_node = 8*2/8 = 2. Aggregated (prefill num-worker 1 + +# decode num-worker 0, RECIPES.md section 5) — the single worker serves both +# phases, so no P/D KV transfer. Dedicated kimi-k3 vLLM bring-up image with +# VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION=1 and the kimi_k3 tool-call/reasoning +# parsers. +# Recipe: benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-agentic.yaml +kimik3-fp4-b200-dynamo-vllm-agentic: + image: vllm/vllm-openai:kimi-k3 + model: moonshotai/Kimi-K3 + model-prefix: kimik3 + runner: cluster:b200-dgxc + precision: fp4 + # framework stays dynamo-vllm for launcher routing, but this variant serves + # DIRECTLY with vllm serve (srt-slurm PR #278 frontend.type: vllm + the + # InferenceX multinode patch) — no dynamo frontend/worker/router involved. + framework: dynamo-vllm + multinode: true + disagg: false + scenarios: + agentic-coding: + # Variant E: SimpleCPUOffloadConnector KV offload to CPU DRAM. + # dram-utilization 0.80 of the capped node DRAM = 2400 GB/node budget; + # the recipe hardcodes the resulting 300 GB/rank connector config. + - dram-utilization: 0.80 + search-space: + - spec-decoding: none + kv-offloading: dram + kv-offload-backend: { name: vllm-simple, version: "kimi-k3" } + conc-list: [1, 8, 16, 32] + prefill: + num-worker: 1 + tp: 8 + pp: 2 + ep: 1 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-agentic.yaml" + # The aggregate worker also performs decode; keep the decode worker + # count at zero so result aggregation counts the 16 GPUs only once. + decode: + num-worker: 0 + tp: 8 + pp: 2 + ep: 1 + dp-attn: false diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 0524d0c0c9..fecd62735b 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5116,3 +5116,18 @@ description: - "Bump image from lmsysorg/sglang:v0.5.14-rocm720-mi35x to lmsysorg/sglang-rocm:v0.5.16-rocm720-mi35x-20260726" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2349 + +- config-keys: + - kimik3-fp4-b200-dynamo-vllm-agentic + description: + - "Add Kimi-K3 MXFP4 B200 aggregated multinode Dynamo-vLLM agentic-coding bring-up (new model on B200; first kimik3 benchmark config)" + - "Aggregated TP8 x PP2 across 2 B200 nodes (16 GPUs), plain TP (NOT TEP: ep 1, no enable-expert-parallel) — the native MXFP4 checkpoint (2.8T total params, ~1.4TB weights) does not fit one 8xB200 node, so TP8 shards attention/dense and PP2 splits the 93 layers. Aggregated mode (prefill num-worker 1 + decode num-worker 0, RECIPES.md section 5): one worker serves prefill and decode, no P/D KV transfer" + - "Dedicated bring-up image vllm/vllm-openai:kimi-k3 with VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION=1 (fuses the K3 LatentMoE tail path), --load-format fastsafetensors, --moe-backend auto, --gpu-memory-utilization 0.90 (0.95 OOMs: the flashinfer trtllm MXFP4 MoE kernel allocates a ~1.6 GiB runtime workspace outside vLLM's pool on the first forward), --no-enable-flashinfer-autotune, --trust-remote-code, --enable-auto-tool-choice, --tool-call-parser kimi_k3, --reasoning-parser kimi_k3 (agentic experiment Variant D: the OpenAI-frontend parser flags are legitimate here because serving is DIRECT vllm serve, not the dynamo worker entrypoint that rejects them)" + - "No explicit max-model-len (vLLM derives the native 1M window from the model config; K3's KDA layers keep per-token KV small — only the 24 gated-MLA layers hold cache), prefix caching on for trajectory reuse, max-num-batched-tokens 8192 so a single long prefill cannot OOM a pipeline stage; conc 1/8/16/32" + - "DIRECT vLLM serving via srt-slurm PR #278 (kylliang/direct-aggregate-vllm, frontend.type: vllm): vllm serve owns the OpenAI port itself, removing the dynamo frontend/worker/router entirely (dynamo install: false) and with it the kimi_k3 tiktoken tokenizer gap that 404'd every request on dynamo <=1.2.1. PR #278 validates single-node only, so patches/srt-slurm-pr278-direct-vllm-multinode.patch extends it to vLLM-native multi-node serve (--master-addr/--nnodes/--node-rank, headless non-leader ranks) for the 2-node TP8xPP2 topology" + - "Model pre-staged at /lustre/fsw/models/Kimi-K3 (moonshotai/Kimi-K3); launcher launch_b200-dgxc.sh gains the kimik3/fp4 model-path mapping, pins the agentic srt-slurm base to upstream NVIDIA/srt-slurm v1.0.36 (validated in #2302/#2341; replaces the cquil11/srt-slurm-nv fork branch whose older srtctl schema rejects newer recipe fields such as benchmark.aiperf_server_metrics), overlays the kimi-k3 agentic recipes onto the clone, and adds the agentic default_mounts (/aiperf_mmap_cache, /hf_hub_cache) already used by the GB200/GB300 agentic paths" + - "aiperf conv-aware routing disabled (AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING=0, same opt-out as the GB300 aggregate AgentX recipes): aiperf's nvext.session_control is a removed POC field this dynamo build 400-rejects at warmup (fifth sweep attempt: tokenizer registered, engine served, all 5 warmup requests 400'd); with a single aggregate worker there is no P/D routing to bind" + - "In-container vLLM patch via setup_script kimi-k3-container-deps.sh: the kimi-k3 image's first decode step crashes in the KDA hybrid-state postprocess (vllm/v1/worker/gpu/model_states/mamba_hybrid.py postprocess_state, IndexError: index_fill_(): Expected dtype int64 for index — torch requires an int64 index but the runner passes the int32 idx_mapping; sixth sweep attempt, first warmup request 500s then the model 503s). The patch coerces the index with .long(), is idempotent, and refuses to run if the image layout changed" + - "Agentic experiment Variant E (of the #2359 direct-vllm Variant D): SimpleCPUOffloadConnector KV offload to CPU DRAM via --kv-transfer-config (kv_role kv_both, lazy_offload false, enable_cross_layers_blocks true), cpu_bytes_to_use_per_rank 299,875,000,000 = the framework's 0.80-utilization DRAM budget for cluster:b200-dgxc (2399 GB/node / 8 ranks), PYTHONHASHSEED=42 so identical prefixes hash to identical block keys across ranks, explicit enable-prefix-caching, and kv-offloading dram / kv-offload-backend vllm-simple labeling in the master entry" + - "Recipe: benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-agentic.yaml on the cluster:b200-dgxc pool" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2370 diff --git a/runners/launch_b200-dgxc.sh b/runners/launch_b200-dgxc.sh index a276644575..fe8e518126 100644 --- a/runners/launch_b200-dgxc.sh +++ b/runners/launch_b200-dgxc.sh @@ -72,6 +72,10 @@ elif [[ $MODEL_PREFIX == "minimaxm3" && $PRECISION == "fp4" ]]; then # NVFP4 checkpoint, pre-staged on the b200-dgxc scratch tree. export MODEL_PATH="/scratch/fsw/models/MiniMax-M3-NVFP4" export SRT_SLURM_MODEL_PREFIX="minimax-m3-nvfp4" +elif [[ $MODEL_PREFIX == "kimik3" && $PRECISION == "fp4" ]]; then + # Native MXFP4 checkpoint, pre-staged on the SRE-managed Lustre tree. + export MODEL_PATH="/lustre/fsw/models/Kimi-K3" + export SRT_SLURM_MODEL_PREFIX="kimik3" else echo "Unsupported model prefix/precision: $MODEL_PREFIX/$PRECISION" echo "Available models under /lustre/fsw/models:" @@ -104,9 +108,25 @@ if [[ "$IS_MULTINODE" == "true" ]]; then fi # TODO(CJQ): make first class upon srt-slurm upstream refactor - if [[ "$IS_AGENTIC" == "1" ]]; then - git clone --branch cam/sa-submission-q2-2026 --single-branch https://github.com/cquil11/srt-slurm-nv.git "$SRT_REPO_DIR" + if [[ "$IS_AGENTIC" == "1" && $MODEL_PREFIX == "kimik3" ]]; then + # Direct-vLLM agentic experiment (Variant D): srt-slurm PR #278 + # (kylliang/direct-aggregate-vllm) adds frontend.type: vllm — `vllm + # serve` owns the OpenAI port itself, no Dynamo layer. The InferenceX + # patch below extends it from single-node to vLLM-native multi-node + # serve (--master-addr/--nnodes/--node-rank + headless non-leader + # ranks) so the 2-node TP8xPP2 topology can run. + git clone --branch kylliang/direct-aggregate-vllm --single-branch https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" || exit 1 cd "$SRT_REPO_DIR" || exit 1 + git apply "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/patches/srt-slurm-pr278-direct-vllm-multinode.patch" || exit 1 + if [[ $MODEL_PREFIX == "kimik3" ]]; then + mkdir -p recipes/vllm/kimi-k3/agentic || exit 1 + cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic" \ + recipes/vllm/kimi-k3/agentic || exit 1 + # In-container vLLM patch for the kimi-k3 image, referenced by the + # recipes' setup_script field (srt-slurm mounts configs/ at /configs). + cp "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/configs/kimi-k3-container-deps.sh" \ + configs/kimi-k3-container-deps.sh || exit 1 + fi elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "dsv4" ]]; then git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" cd "$SRT_REPO_DIR" || exit 1 @@ -207,6 +227,22 @@ if [[ "$IS_MULTINODE" == "true" ]]; then export OSL="$OSL" export EVAL_ONLY="${EVAL_ONLY:-false}" + # Agentic runs bind-mount two persistent caches into every worker + # container (Lustre, shared across nodes): aiperf's content-addressed + # dataset mmap cache and the HF hub cache holding the trace dataset + # download. The container-side paths are referenced by the agentic + # recipes' benchmark.env (AIPERF_DATASET_MMAP_CACHE_DIR=/aiperf_mmap_cache, + # HF_HUB_CACHE=/hf_hub_cache). + DEFAULT_MOUNTS_BLOCK="" + if [[ "$IS_AGENTIC" == "1" ]]; then + HF_HUB_CACHE_HOST_PATH="/lustre/fsw/gharunners/hf-hub-cache" + mkdir -p "$AIPERF_MMAP_CACHE_HOST_PATH" "$HF_HUB_CACHE_HOST_PATH" + chmod 777 "$AIPERF_MMAP_CACHE_HOST_PATH" "$HF_HUB_CACHE_HOST_PATH" 2>/dev/null || true + DEFAULT_MOUNTS_BLOCK="default_mounts: + ${AIPERF_MMAP_CACHE_HOST_PATH}: /aiperf_mmap_cache + ${HF_HUB_CACHE_HOST_PATH}: /hf_hub_cache" + fi + # Create srtslurm.yaml for srtctl (used by both frameworks) SRTCTL_ROOT="${GITHUB_WORKSPACE}/${SRT_REPO_DIR}" echo "Creating srtslurm.yaml configuration..." @@ -234,6 +270,7 @@ containers: "${IMAGE}": "${SQUASH_FILE}" nginx-sqsh: "${NGINX_SQUASH_FILE}" use_exclusive_sbatch_directive: true +${DEFAULT_MOUNTS_BLOCK} EOF echo "Generated srtslurm.yaml:"