From f42a66a7e3fbe919b8b58183f0afae286cf7b919 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Wed, 12 Aug 2026 12:10:01 -0500 Subject: [PATCH 1/4] perf(agentx): refresh DSV4 B300 SGLang MTP sweep --- .../agentic/dsv4_fp4_b300_sglang_mtp.sh | 263 ++++++++++++++++++ configs/nvidia-master.yaml | 18 ++ perf-changelog.yaml | 10 + 3 files changed, 291 insertions(+) create mode 100755 benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh diff --git a/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh new file mode 100755 index 000000000..8c2658c9e --- /dev/null +++ b/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh @@ -0,0 +1,263 @@ +#!/usr/bin/env bash +set -eo pipefail +set -x + +# Agentic trace replay for DeepSeek-V4-Pro FP4 on B300 with native EAGLE MTP. +# Throughput uses the committed golden synthetic AL; eval retains real target +# verification. +# +# KV_OFFLOADING=dram requires KV_OFFLOAD_BACKEND=hicache. + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +INFERENCEX_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)" +export INFMAX_CONTAINER_WORKSPACE="${INFMAX_CONTAINER_WORKSPACE:-/workspace}" + +# The B200 DeepSeek-V4 Blackwell image installs SGLang editable under +# /workspace, so its launcher mounts InferenceX at /ix instead. Resolve the +# agentic tooling and results against the actual repository mount so the image +# can keep its /workspace install and GitHub Actions can collect the outputs. +if [[ ! -d "$INFMAX_CONTAINER_WORKSPACE/utils/aiperf" ]]; then + export INFMAX_CONTAINER_WORKSPACE="$INFERENCEX_ROOT" +fi +if [[ "${RESULT_DIR:-}" == /workspace/* && "$INFMAX_CONTAINER_WORKSPACE" != /workspace ]]; then + export RESULT_DIR="$INFMAX_CONTAINER_WORKSPACE/${RESULT_DIR#/workspace/}" +fi +source "$INFERENCEX_ROOT/benchmarks/benchmark_lib.sh" + +export AIPERF_REQUIRED_SERVER_METRIC_PREFIX="sglang:" + +check_env_vars MODEL TP CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION EP_SIZE DP_ATTENTION + +if [[ -n "${SLURM_JOB_ID:-}" ]]; then + echo "JOB $SLURM_JOB_ID running on ${SLURMD_NODENAME:-unknown}" +fi + +if [[ -n "${MODEL_PATH:-}" ]]; then + if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then + hf download "$MODEL" --local-dir "$MODEL_PATH" + fi +else + hf download "$MODEL" + export MODEL_PATH="$MODEL" +fi +nvidia-smi + +resolve_trace_source + +# Keep AIPerf's Transformers-main dependency from replacing the older +# Transformers build pinned by the B200-specialized SGLang image. The server +# always launches with the image's original interpreter; AIPerf and result +# processing use the isolated environment when InferenceX is mounted at /ix. +SGLANG_PYTHON="$(command -v python3)" +if [[ "$INFMAX_CONTAINER_WORKSPACE" != /workspace ]]; then + AGENTIC_VENV="${AGENTIC_VENV:-/tmp/inferencex-agentic-venv}" + "$SGLANG_PYTHON" -m venv "$AGENTIC_VENV" + export PATH="$AGENTIC_VENV/bin:$PATH" +fi +install_agentic_deps + +SERVER_LOG="$RESULT_DIR/server.log" +mkdir -p "$RESULT_DIR" + +export SGLANG_ENABLE_UNIFIED_RADIX_TREE=1 +export SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS=1 + +CACHE_ARGS=() +if require_agentic_kv_offload_backend hicache; then + # DeepSeek V4 HiCache currently rejects --hicache-size and supports + # capacity control only through a host/device token-capacity ratio. + # DSv4 exposes capacity as a host/device token ratio rather than bytes. + # Measurements put TP8 ratio=2 near 950 GB and TP4 ratio=8 near 1 TB, + # both below their configured capacities. The old TP4 ratio=16 + # used roughly 2 TB and violated the half-node allocation rule. + if [ "$TP" -ge 8 ]; then + DEFAULT_HICACHE_RATIO=2 + else + DEFAULT_HICACHE_RATIO=8 + fi + HICACHE_RATIO="${HICACHE_RATIO:-$DEFAULT_HICACHE_RATIO}" + if [ "$HICACHE_RATIO" -gt "$DEFAULT_HICACHE_RATIO" ]; then + echo "Error: HICACHE_RATIO=$HICACHE_RATIO exceeds configured limit $DEFAULT_HICACHE_RATIO" >&2 + exit 1 + fi + HICACHE_WRITE_POLICY="${HICACHE_WRITE_POLICY:-write_back}" + HICACHE_IO_BACKEND="${HICACHE_IO_BACKEND:-direct}" + HICACHE_MEM_LAYOUT="${HICACHE_MEM_LAYOUT:-page_first_direct}" + CACHE_ARGS=( + --enable-hierarchical-cache + --hicache-ratio "$HICACHE_RATIO" + --hicache-write-policy "$HICACHE_WRITE_POLICY" + --hicache-io-backend "$HICACHE_IO_BACKEND" + --hicache-mem-layout "$HICACHE_MEM_LAYOUT" + ) + echo "HiCache DSv4 CPU tier: ratio=$HICACHE_RATIO, capacity=${TOTAL_CPU_DRAM_GB} GB, write_policy=$HICACHE_WRITE_POLICY, io_backend=$HICACHE_IO_BACKEND, mem_layout=$HICACHE_MEM_LAYOUT" +fi + +USE_SGLANG_ROUTER=false +SGLANG_BACKEND_PORT="$PORT" +ROUTER_LOG="$RESULT_DIR/router.log" +if [ "$DP_ATTENTION" = "true" ]; then + USE_SGLANG_ROUTER=true + export AIPERF_HTTP_X_SMG_ROUTING_KEY_FROM_CORRELATION_ID=true + SGLANG_BACKEND_PORT=$((PORT + 1)) + SGLANG_ROUTER_METRICS_PORT=$((PORT + 10000)) + SGLANG_ROUTER_CMD=("$SGLANG_PYTHON" -m sglang_router.launch_router) +fi + +PARALLEL_ARGS=(--tp "$TP") +METRICS_ARGS=(--enable-metrics --enable-cache-report) +MEM_FRACTION_STATIC=0.88 +CHUNKED_PREFILL_SIZE=8192 +if [ "$DP_ATTENTION" = "true" ]; then + PARALLEL_ARGS+=( + --dp "$TP" + --tokenizer-worker-num "$TP" + --enable-dp-attention + --enable-dp-attention-local-control-broadcast + --incremental-streaming-output + --stream-interval 20 + --dist-init-addr "127.0.0.1:$((PORT + 2000))" + --ep-size "$EP_SIZE" + --moe-runner-backend flashinfer_mxfp4 + --disable-flashinfer-autotune + ) + MEM_FRACTION_STATIC=0.95 + CHUNKED_PREFILL_SIZE=16384 +else + PARALLEL_ARGS+=( + --moe-runner-backend flashinfer_mxfp4 + --disable-flashinfer-autotune + ) +fi + +MODEL_ARGS=( + --attention-backend compressed + --page-size 256 + --disable-shared-experts-fusion +) + +# AgentX concurrency counts live session trees, not individual requests. +# Allow subagent fan-out to exceed CONC without clipping request bursts. +MAX_RUNNING_REQUESTS=$((2 * CONC)) +CUDA_GRAPH_MAX_BS=$CONC +[ "$CUDA_GRAPH_MAX_BS" -gt 64 ] && CUDA_GRAPH_MAX_BS=64 + +export PYTHONNOUSERSITE=1 +export TORCH_CUDA_ARCH_LIST=10.0 +# Agentic warmup dispatches hundreds of large prompts at once. SGLang's +# tokenizer process can leave request bytes unacknowledged for longer than +# AIPerf's 30-second TCP_USER_TIMEOUT while it admits that initial burst, +# causing Linux to abort otherwise-live localhost connections. Keep the +# six-hour request timeout unchanged, but allow up to 15 minutes for TCP +# progress before declaring the connection dead. +export AIPERF_HTTP_TCP_USER_TIMEOUT=900000 +export SGLANG_JIT_DEEPGEMM_FAST_WARMUP=1 +export SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT=1 +export SGLANG_OPT_USE_JIT_NORM=1 +export SGLANG_OPT_USE_JIT_INDEXER_METADATA=1 +export SGLANG_OPT_USE_TOPK_V2=1 +export SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2=1 +if [ "${EVAL_ONLY}" != "true" ]; then + export SGLANG_SIMULATE_ACC_LEN=2.49 + export SGLANG_SIMULATE_ACC_METHOD=match-expected + export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token +fi +TRITON_PTXAS_PATH=$(find \ + /usr/local/cuda* \ + /usr/local/lib/python*/dist-packages/nvidia \ + /usr/local/lib/python*/site-packages/nvidia \ + -type f -name ptxas -perm -u+x -print -quit 2>/dev/null || true) +if [ -n "$TRITON_PTXAS_PATH" ]; then + export TRITON_PTXAS_PATH + echo "Using ptxas for Triton: $TRITON_PTXAS_PATH" +fi +SGLANG_CMD=( + "$SGLANG_PYTHON" -m sglang.launch_server + --model-path "$MODEL_PATH" + --served-model-name "$MODEL" + --host 0.0.0.0 + --port "$SGLANG_BACKEND_PORT" + --trust-remote-code + "${PARALLEL_ARGS[@]}" + --mem-fraction-static "$MEM_FRACTION_STATIC" + --swa-full-tokens-ratio 0.1 + --max-running-requests "$MAX_RUNNING_REQUESTS" + --cuda-graph-max-bs "$CUDA_GRAPH_MAX_BS" + --allow-auto-truncate + --chunked-prefill-size "$CHUNKED_PREFILL_SIZE" + --tool-call-parser deepseekv4 + --reasoning-parser deepseek-v4 + --chat-template "$SCRIPT_DIR/../chat_templates/deepseek_v4_thinking.jinja" + --watchdog-timeout 1800 + --speculative-algorithm EAGLE + --speculative-num-steps 3 + --speculative-eagle-topk 1 + --speculative-num-draft-tokens 4 + "${MODEL_ARGS[@]}" + "${METRICS_ARGS[@]}" + "${CACHE_ARGS[@]}" +) + +write_command "$RESULT_DIR/sglang_command.txt" "${SGLANG_CMD[@]}" + +{ + echo "=== SGLANG_* env vars at launch ===" + env | grep -E '^SGLANG_' | sort + echo "===================================" +} | tee "$SERVER_LOG" + +echo "Starting SGLang server for B300..." +"${SGLANG_CMD[@]}" >> "$SERVER_LOG" 2>&1 & +SERVER_PID=$! +echo "Server PID: $SERVER_PID" + +capture_cache_metrics() { + { + echo "=== SGLang cache metrics snapshot $(date --iso-8601=seconds) ===" + curl -fsS "http://localhost:$SGLANG_BACKEND_PORT/metrics" 2>/dev/null \ + | grep -E '^(sglang:(cache_hit_rate|cached_tokens_total|prompt_tokens_total|hicache_host_used_tokens|hicache_host_total_tokens|token_usage|num_requests_running|num_requests_waiting))' \ + || true + echo "============================================================" + } >> "$SERVER_LOG" +} + +wait_for_ready \ + --endpoint "http://localhost:$SGLANG_BACKEND_PORT/health" \ + --log "$SERVER_LOG" \ + --pid "$SERVER_PID" + +if [ "$USE_SGLANG_ROUTER" = "true" ]; then + echo "Starting SGLang router on port $PORT for $TP DP ranks..." + "${SGLANG_ROUTER_CMD[@]}" \ + --worker-urls "http://localhost:$SGLANG_BACKEND_PORT" \ + --policy consistent_hashing \ + --request-id-headers x-correlation-id \ + --dp-aware \ + --host 0.0.0.0 \ + --port "$PORT" \ + --prometheus-host 127.0.0.1 \ + --prometheus-port "$SGLANG_ROUTER_METRICS_PORT" \ + --connect-timeout-secs 900 \ + --request-timeout-secs 14400 \ + --disable-health-check \ + --disable-retries > "$ROUTER_LOG" 2>&1 & + ROUTER_PID=$! + echo "Router PID: $ROUTER_PID" + wait_for_ready \ + --endpoint "http://localhost:$PORT/health" \ + --log "$ROUTER_LOG" \ + --pid "$ROUTER_PID" +fi + +if [ "${#METRICS_ARGS[@]}" -gt 0 ]; then + capture_cache_metrics + trap capture_cache_metrics EXIT +fi + +if [ "${EVAL_ONLY}" = "true" ]; then + run_eval --port "$PORT" +else + build_replay_cmd "$RESULT_DIR" + REPLAY_CMD+=" --server-metrics http://localhost:$SGLANG_BACKEND_PORT/metrics" + run_agentic_replay_and_write_outputs "$RESULT_DIR" +fi diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 757a15c0e..7b4f1472b 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -1142,6 +1142,24 @@ dsv4-fp4-b300-sglang: - { tp: 8, ep: 8, dp-attn: true, conc-start: 2048, conc-end: 2048 } - { tp: 8, ep: 8, dp-attn: true, conc-start: 4096, conc-end: 4096 } +dsv4-fp4-b300-sglang-agentic-hicache-mtp: + image: lmsysorg/sglang:v0.5.17-cu130 + model: deepseek-ai/DeepSeek-V4-Pro + model-prefix: dsv4 + runner: cluster:b300-nv + precision: fp4 + framework: sglang + multinode: false + scenarios: + agentic-coding: + - dram-utilization: 0.80 + search-space: + - { tp: 4, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 4, 8, 16, 20, 24, 32] } + - { tp: 8, kv-offloading: none, spec-decoding: mtp, conc-list: [1, 4, 8, 16, 32, 40, 48, 52, 56, 60, 64, 72] } + - { tp: 4, ep: 4, dp-attn: true, kv-offloading: none, spec-decoding: mtp, conc-list: [8, 16, 24, 32, 40, 64], router: { name: sglang-router, version: "0.3.2" } } + - { tp: 4, ep: 4, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, spec-decoding: mtp, conc-list: [32, 40, 48, 56, 64, 72, 80, 88, 96, 128], router: { name: sglang-router, version: "0.3.2" } } + - { tp: 8, ep: 8, dp-attn: true, kv-offloading: none, spec-decoding: mtp, conc-list: [52, 72, 100, 128, 144, 196, 512], router: { name: sglang-router, version: "0.3.2" } } + # DeepSeek-V4-Pro on B300 with EAGLE/MTP speculative decoding. Recipe is # selected inside benchmarks/single_node/dsv4_fp4_b300_sglang_mtp.sh by # DP_ATTENTION: diff --git a/perf-changelog.yaml b/perf-changelog.yaml index a1875d61b..b0101f08a 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5438,6 +5438,16 @@ - "Search space is one TP8 GPU-resident arm (kv-offloading none) at conc-list [4, 8, 10, 12, 14, 16]; the LMCache/vllm-native/SimpleCPUOffload DRAM-offload backends are wired in the recipe but off this sweep's ladder." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2508 +- config-keys: + - dsv4-fp4-b300-sglang-agentic-hicache-mtp + scenario-type: + - agentic-coding + description: + - "Refresh the purged B300 DeepSeek-V4-Pro SGLang AgentX TP4, TP8, DP-attention, and HiCache grid on SGLang v0.5.17." + - "Use native EAGLE MTP (3 steps, top-k 1, 4 draft tokens) and golden synthetic acceptance length 2.49 for throughput; eval retains real verification." + - "Follow the official SGLang DeepSeek-V4 Blackwell recipe and require nonempty SGLang server metrics." + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2577 + - config-keys: - minimaxm3-fp4-b200-dynamo-vllm-mtp description: From 38171d3944698fa57237bbad263d6a4573e8be7b Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Wed, 12 Aug 2026 12:54:04 -0500 Subject: [PATCH 2/4] fix: allow B300 eval result metadata --- benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh | 1 + 1 file changed, 1 insertion(+) diff --git a/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh index 8c2658c9e..a0b29fd86 100755 --- a/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh @@ -255,6 +255,7 @@ if [ "${#METRICS_ARGS[@]}" -gt 0 ]; then fi if [ "${EVAL_ONLY}" = "true" ]; then + git config --global --add safe.directory "$INFMAX_CONTAINER_WORKSPACE" run_eval --port "$PORT" else build_replay_cmd "$RESULT_DIR" From b3f7b4990307d7d8f57c92eb9c6fe5af200d6e7f Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Wed, 12 Aug 2026 20:03:22 -0500 Subject: [PATCH 3/4] fix: stabilize B300 DeepSeek V4 AgentX sweep --- .../single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh | 12 ++++++++++++ perf-changelog.yaml | 2 +- 2 files changed, 13 insertions(+), 1 deletion(-) diff --git a/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh b/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh index a0b29fd86..fc917052c 100755 --- a/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh +++ b/benchmarks/single_node/agentic/dsv4_fp4_b300_sglang_mtp.sh @@ -63,6 +63,7 @@ export SGLANG_ENABLE_UNIFIED_RADIX_TREE=1 export SGLANG_OPT_UNIFIED_CACHE_FREE_OUT_OF_WINDOW_SLOTS=1 CACHE_ARGS=() +WARMUP_ARGS=() if require_agentic_kv_offload_backend hicache; then # DeepSeek V4 HiCache currently rejects --hicache-size and supports # capacity control only through a host/device token-capacity ratio. @@ -90,6 +91,9 @@ if require_agentic_kv_offload_backend hicache; then --hicache-io-backend "$HICACHE_IO_BACKEND" --hicache-mem-layout "$HICACHE_MEM_LAYOUT" ) + # AIPerf owns the representative warmup for AgentX. Avoid SGLang's + # redundant per-DP warmup timing out after the API is already healthy. + WARMUP_ARGS=(--skip-server-warmup) echo "HiCache DSv4 CPU tier: ratio=$HICACHE_RATIO, capacity=${TOTAL_CPU_DRAM_GB} GB, write_policy=$HICACHE_WRITE_POLICY, io_backend=$HICACHE_IO_BACKEND, mem_layout=$HICACHE_MEM_LAYOUT" fi @@ -122,6 +126,10 @@ if [ "$DP_ATTENTION" = "true" ]; then --disable-flashinfer-autotune ) MEM_FRACTION_STATIC=0.95 + if [ "$CONC" -ge 512 ]; then + # Leave room for FlashInfer's transient MoE workspace at the DEP8 tail. + MEM_FRACTION_STATIC=0.94 + fi CHUNKED_PREFILL_SIZE=16384 else PARALLEL_ARGS+=( @@ -151,6 +159,9 @@ export TORCH_CUDA_ARCH_LIST=10.0 # six-hour request timeout unchanged, but allow up to 15 minutes for TCP # progress before declaring the connection dead. export AIPERF_HTTP_TCP_USER_TIMEOUT=900000 +# Outlast AIPerf's pooled connections so an inter-turn idle gap cannot race +# Uvicorn's five-second keep-alive closure. +export SGLANG_TIMEOUT_KEEP_ALIVE=900 export SGLANG_JIT_DEEPGEMM_FAST_WARMUP=1 export SGLANG_OPT_SWA_SPLIT_LEAF_ON_INSERT=1 export SGLANG_OPT_USE_JIT_NORM=1 @@ -196,6 +207,7 @@ SGLANG_CMD=( "${MODEL_ARGS[@]}" "${METRICS_ARGS[@]}" "${CACHE_ARGS[@]}" + "${WARMUP_ARGS[@]}" ) write_command "$RESULT_DIR/sglang_command.txt" "${SGLANG_CMD[@]}" diff --git a/perf-changelog.yaml b/perf-changelog.yaml index b0101f08a..cc9325792 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5445,7 +5445,7 @@ description: - "Refresh the purged B300 DeepSeek-V4-Pro SGLang AgentX TP4, TP8, DP-attention, and HiCache grid on SGLang v0.5.17." - "Use native EAGLE MTP (3 steps, top-k 1, 4 draft tokens) and golden synthetic acceptance length 2.49 for throughput; eval retains real verification." - - "Follow the official SGLang DeepSeek-V4 Blackwell recipe and require nonempty SGLang server metrics." + - "Follow the official SGLang DeepSeek-V4 Blackwell recipe, require nonempty SGLang server metrics, keep pooled AgentX connections alive, let AIPerf own HiCache warmup, and reserve transient MoE workspace at DEP8 c512." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2577 - config-keys: From 0aac240dc694d86b16a9c6dcefcf94afb0fb13ac Mon Sep 17 00:00:00 2001 From: Cameron Quilici Date: Thu, 13 Aug 2026 15:10:00 -0500 Subject: [PATCH 4/4] Update perf-changelog.yaml --- perf-changelog.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 8f17211c1..34e463000 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5928,3 +5928,4 @@ - "Use native EAGLE MTP (3 steps, top-k 1, 4 draft tokens) and golden synthetic acceptance length 2.49 for throughput; eval retains real verification." - "Follow the official SGLang DeepSeek-V4 Blackwell recipe, require nonempty SGLang server metrics, keep pooled AgentX connections alive, let AIPerf own HiCache warmup, and reserve transient MoE workspace at DEP8 c512." pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2577 +