Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
74 commits
Select commit Hold shift + click to select a range
52d7a56
feat: add Qwen3.5 FP4 B200 AgentX MTP
cquil11 Jul 30, 2026
bf03912
chore: trigger B200 AgentX sweep
cquil11 Jul 30, 2026
9c682e2
perf: expand B200 AgentX preflight range
cquil11 Jul 30, 2026
2d6c12b
chore: use SGLang v0.5.16 for Qwen B200
cquil11 Jul 30, 2026
b118256
chore: collect SGLang cache metrics on B200
cquil11 Jul 30, 2026
e2c4ab5
fix: remove unsupported SGLang tool-choice flag on B200
cquil11 Jul 30, 2026
4a587bd
perf: prefer local Qwen FP4 weights on B200
cquil11 Jul 30, 2026
1b863c5
perf: prepare Qwen AgentX cache and DEP tuning
cquil11 Jul 30, 2026
1e89789
perf: add Qwen B200 DEP probes
cquil11 Jul 30, 2026
f7e3b37
perf: add matched B200 HiCache cliff probe
cquil11 Jul 30, 2026
191ebbe
perf: isolate B200 HiCache validation point
cquil11 Jul 30, 2026
b439bf7
perf: extend Qwen B200 TP4 cliff probe
cquil11 Jul 30, 2026
df5a388
perf: drop dominated Qwen B200 DEP arm
cquil11 Jul 30, 2026
ce05286
fix: enforce Qwen HiCache DRAM total
cquil11 Jul 30, 2026
718ff2b
perf: add B200 TP4 c72 HiCache A/B
cquil11 Jul 30, 2026
e7af32b
chore: merge current main into B200 submission
cquil11 Jul 30, 2026
efbe79d
refactor: remove dominated B200 DEP path
cquil11 Jul 30, 2026
0eddcce
perf: densify B200 TP2 cache cliff
cquil11 Jul 30, 2026
123f0f8
perf: resolve B200 Qwen cache cliff
cquil11 Jul 30, 2026
b2477e4
perf: prune dominated B200 TP2 point
cquil11 Jul 30, 2026
3ecfbc8
perf: stop B200 TP4 before cache thrashing
cquil11 Jul 30, 2026
e518bc0
perf: switch B200 TP2 to HiCache after c14
cquil11 Jul 30, 2026
a7b78e1
perf: prune dominated B200 TP4 c66
cquil11 Jul 30, 2026
28ac3e6
fix(agentx): use stable tokenizer startup
cquil11 Jul 30, 2026
801f922
perf(agentx): cap B200 no-offload frontier
cquil11 Jul 30, 2026
8d05af8
fix(agentx): scale Qwen tokenization by topology
cquil11 Jul 30, 2026
3ed39d3
perf(agentx): trim B200 cache-thrash tail
cquil11 Jul 30, 2026
3a1b5f7
fix(agentx): cap Qwen trace idle gaps
cquil11 Jul 30, 2026
e1ab4ff
chore(aiperf): pin merged trace idle cap branch
cquil11 Jul 30, 2026
6d21400
chore(agentx): bump AIPerf trace idle cap fix
cquil11 Jul 30, 2026
4d63a16
fix(agentx): update AIPerf trace-cap cleanup
cquil11 Jul 30, 2026
920618b
chore(agentx): pin reconstruction-only idle cap
cquil11 Jul 30, 2026
c382acf
Merge origin/main into agent/qwen35-fp4-b200-agentx-mtp
cquil11 Jul 30, 2026
e040e7c
fix(agentx): update AIPerf warmup handoff
cquil11 Jul 30, 2026
2aa9415
fix(agentx): retain baseline parents across warmup
cquil11 Jul 30, 2026
0b89cdf
chore: merge main into B200 AgentX branch
cquil11 Jul 30, 2026
cfb86f1
fix(agentx): extend SGLang keep-alive
cquil11 Jul 31, 2026
1b4d774
fix(agentx): pin validated AIPerf scheduling on B200
cquil11 Jul 31, 2026
56462c2
chore: merge main into Qwen B200 AgentX branch
cquil11 Aug 2, 2026
5e3d346
test(agentx): validate current AIPerf on Qwen B200
cquil11 Aug 2, 2026
56fae60
feat: add Qwen3.5 FP4 B200 AgentX MTP
cquil11 Jul 30, 2026
a46d329
chore: trigger B200 AgentX sweep
cquil11 Jul 30, 2026
c6a2a55
perf: expand B200 AgentX preflight range
cquil11 Jul 30, 2026
9a22637
chore: use SGLang v0.5.16 for Qwen B200
cquil11 Jul 30, 2026
26c69f1
chore: collect SGLang cache metrics on B200
cquil11 Jul 30, 2026
1246b0d
fix: remove unsupported SGLang tool-choice flag on B200
cquil11 Jul 30, 2026
96dde6d
perf: prefer local Qwen FP4 weights on B200
cquil11 Jul 30, 2026
7e428e9
perf: prepare Qwen AgentX cache and DEP tuning
cquil11 Jul 30, 2026
8950545
perf: add Qwen B200 DEP probes
cquil11 Jul 30, 2026
6d63024
perf: add matched B200 HiCache cliff probe
cquil11 Jul 30, 2026
4350f4f
perf: isolate B200 HiCache validation point
cquil11 Jul 30, 2026
62ff1d5
perf: extend Qwen B200 TP4 cliff probe
cquil11 Jul 30, 2026
419dbbd
perf: drop dominated Qwen B200 DEP arm
cquil11 Jul 30, 2026
98b368b
fix: enforce Qwen HiCache DRAM total
cquil11 Jul 30, 2026
dbcc777
perf: add B200 TP4 c72 HiCache A/B
cquil11 Jul 30, 2026
eec9d6a
refactor: remove dominated B200 DEP path
cquil11 Jul 30, 2026
2fc6fa4
perf: densify B200 TP2 cache cliff
cquil11 Jul 30, 2026
0601efa
perf: resolve B200 Qwen cache cliff
cquil11 Jul 30, 2026
e9357bf
perf: prune dominated B200 TP2 point
cquil11 Jul 30, 2026
a1eac76
perf: stop B200 TP4 before cache thrashing
cquil11 Jul 30, 2026
2e2b7dd
perf: switch B200 TP2 to HiCache after c14
cquil11 Jul 30, 2026
4c8567c
perf: prune dominated B200 TP4 c66
cquil11 Jul 30, 2026
6620168
fix(agentx): use stable tokenizer startup
cquil11 Jul 30, 2026
03fa56d
perf(agentx): cap B200 no-offload frontier
cquil11 Jul 30, 2026
6de7c9f
fix(agentx): scale Qwen tokenization by topology
cquil11 Jul 30, 2026
737115c
perf(agentx): trim B200 cache-thrash tail
cquil11 Jul 30, 2026
a5fc138
fix(agentx): cap Qwen trace idle gaps
cquil11 Jul 30, 2026
360a4d9
fix(agentx): extend SGLang keep-alive
cquil11 Jul 31, 2026
1267ca9
test(agentx): validate current AIPerf on Qwen B200
cquil11 Aug 2, 2026
8759ed8
Update nvidia-master.yaml
cquil11 Aug 3, 2026
d4eee1e
chore: re-anchor Qwen B200 validated sweep
cquil11 Aug 3, 2026
ad7465d
Merge branch 'main' into agent/qwen35-fp4-b200-agentx-mtp
cquil11 Aug 3, 2026
90e2985
Update perf-changelog.yaml
cquil11 Aug 3, 2026
1e93ef9
chore(b200): remove unused Qwen NVMe probe
cquil11 Aug 3, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
178 changes: 178 additions & 0 deletions benchmarks/single_node/agentic/qwen3.5_fp4_b200_sglang_mtp.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,178 @@
#!/usr/bin/env bash
set -euo pipefail
set -x

# AgentX trace replay for Qwen3.5-397B-A17B NVFP4 on B200 with SGLang
# native NEXTN MTP. Throughput uses the committed golden synthetic AL; evals
# retain real target-model verification.

source "$(dirname "$0")/../../benchmark_lib.sh"

# Use the lightweight GSM8K eval instead of the AgentX SWE-bench default.
export EVAL_FRAMEWORK="lm-eval"

check_env_vars \
MODEL TP CONC EP_SIZE KV_OFFLOADING \
TOTAL_CPU_DRAM_GB RESULT_DIR DURATION

SCHEDULER_RECV_INTERVAL=${SCHEDULER_RECV_INTERVAL:-10}

if [[ -n "${SLURM_JOB_ID:-}" ]]; then
echo "JOB $SLURM_JOB_ID running on ${SLURMD_NODENAME:-unknown}"
fi

if [[ -n "${MODEL_PATH:-}" ]]; then
if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then
hf download "$MODEL" --local-dir "$MODEL_PATH"
fi
else
hf download "$MODEL"
export MODEL_PATH="$MODEL"
fi
nvidia-smi

export WEKA_LOADER_OVERRIDE=semianalysis_cc_traces_weka_062126_256k
resolve_trace_source
install_agentic_deps

SERVER_LOG="$RESULT_DIR/server.log"
mkdir -p "$RESULT_DIR"

CACHE_ARGS=()
if require_agentic_kv_offload_backend hicache; then
REQUESTED_HICACHE_TOTAL_GB="${HICACHE_TOTAL_CPU_DRAM_GB:-$TOTAL_CPU_DRAM_GB}"
if [ "$REQUESTED_HICACHE_TOTAL_GB" -gt "$TOTAL_CPU_DRAM_GB" ]; then
echo "Error: requested HiCache pool ${REQUESTED_HICACHE_TOTAL_GB} GB exceeds configured capacity ${TOTAL_CPU_DRAM_GB} GB" >&2
exit 1
fi
TOTAL_CPU_DRAM_GB="$REQUESTED_HICACHE_TOTAL_GB"
# SGLang applies --hicache-size independently to Qwen's target KV and
# Mamba pools. Native NEXTN also creates a draft KV pool with the same
# slot count; its one attention layer adds 1/15 of the target KV bytes.
# Reserve 1 GB/rank for page alignment and enforce H * 31/15 per rank.
HICACHE_ALIGNMENT_RESERVE_GB=$TP
HICACHE_USABLE_TOTAL_GB=$((TOTAL_CPU_DRAM_GB - HICACHE_ALIGNMENT_RESERVE_GB))
if [ "$HICACHE_USABLE_TOTAL_GB" -lt 1 ]; then
echo "Error: insufficient DRAM after HiCache alignment reserve" >&2
exit 1
fi
MAX_HICACHE_SIZE_GB=$((HICACHE_USABLE_TOTAL_GB * 15 / TP / 31))
HICACHE_SIZE_GB="${HICACHE_SIZE_GB:-$MAX_HICACHE_SIZE_GB}"
if [ "$HICACHE_SIZE_GB" -lt 1 ] || [ "$HICACHE_SIZE_GB" -gt "$MAX_HICACHE_SIZE_GB" ]; then
echo "Error: HICACHE_SIZE_GB=$HICACHE_SIZE_GB outside 1..$MAX_HICACHE_SIZE_GB" >&2
exit 1
fi
PROJECTED_HICACHE_TOTAL_GB=$(((HICACHE_SIZE_GB * TP * 31 + 14) / 15 + HICACHE_ALIGNMENT_RESERVE_GB))
if [ "$PROJECTED_HICACHE_TOTAL_GB" -gt "$TOTAL_CPU_DRAM_GB" ]; then
echo "Error: projected HiCache use ${PROJECTED_HICACHE_TOTAL_GB} GB exceeds configured capacity ${TOTAL_CPU_DRAM_GB} GB" >&2
exit 1
fi
echo "HiCache CPU pools: ${HICACHE_SIZE_GB} GB target + Mamba + 1/15 draft per rank across TP=${TP}; projected node total ${PROJECTED_HICACHE_TOTAL_GB} GB <= ${TOTAL_CPU_DRAM_GB} GB"
CACHE_ARGS=(
--page-size 64
--enable-hierarchical-cache
--hicache-size "$HICACHE_SIZE_GB"
--hicache-io-backend kernel
--hicache-mem-layout page_first
--hicache-write-policy write_through_selective
)
fi

PARALLEL_ARGS=(
--tp "$TP"
--dp 1
--ep-size "$EP_SIZE"
)

# TP4 needs parallel tokenization to keep 256k AgentX warmups below the client
# request timeout. Keep TP2 on SGLang's single-worker default: multi-tokenizer
# startup races with the TP2 HiCache shared-memory initialization path.
TOKENIZER_ARGS=()
if [ "$TP" -ge 4 ]; then
TOKENIZER_ARGS=(--tokenizer-worker-num 6)
fi

# AgentX concurrency counts live session trees rather than individual HTTP
# requests. Leave room for subagent fan-out and avoid spending HBM on graphs
# above the batch sizes that remain useful for this long-context workload.
MAX_RUNNING_REQUESTS=$((2 * CONC))
CUDA_GRAPH_MAX_BS="$CONC"
[ "$CUDA_GRAPH_MAX_BS" -gt 64 ] && CUDA_GRAPH_MAX_BS=64

export TORCH_CUDA_ARCH_LIST="10.0"
export PYTHONNOUSERSITE=1
export NCCL_NVLS_ENABLE=1
export SGL_ENABLE_JIT_DEEPGEMM=false
export SGLANG_ENABLE_FLASHINFER_GEMM=true
# Keep server-side connections alive beyond AIPerf's 300-second client pool
# timeout so bursty AgentX trajectories cannot reuse a closing idle socket.
export SGLANG_TIMEOUT_KEEP_ALIVE=1800

if [ "${EVAL_ONLY:-false}" != "true" ]; then
export SGLANG_SIMULATE_ACC_LEN=3.39
export SGLANG_SIMULATE_ACC_METHOD=match-expected
export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token
fi

SGLANG_CMD=(
python3 -m sglang.launch_server
--model-path "$MODEL_PATH"
--served-model-name "$MODEL"
--host 0.0.0.0
--port "$PORT"
--trust-remote-code
"${PARALLEL_ARGS[@]}"
--enable-symm-mem
--quantization modelopt_fp4
--fp4-gemm-backend flashinfer_cutlass
--kv-cache-dtype fp8_e4m3
--mamba-ssm-dtype bfloat16
--attention-backend trtllm_mha
--moe-runner-backend flashinfer_trtllm
--cuda-graph-max-bs "$CUDA_GRAPH_MAX_BS"
--max-running-requests "$MAX_RUNNING_REQUESTS"
--max-prefill-tokens 16384
--chunked-prefill-size 16384
--mem-fraction-static 0.80
--stream-interval 50
--scheduler-recv-interval "$SCHEDULER_RECV_INTERVAL"
"${TOKENIZER_ARGS[@]}"
--tokenizer-path "$MODEL"
--reasoning-parser qwen3
--tool-call-parser qwen3_coder
--speculative-algorithm NEXTN
--speculative-num-steps 3
--speculative-eagle-topk 1
--speculative-num-draft-tokens 4
--enable-metrics
--enable-cache-report
"${CACHE_ARGS[@]}"
)

printf '%q ' "${SGLANG_CMD[@]}" | tee "$RESULT_DIR/sglang_command.txt"
printf '\n' | tee -a "$RESULT_DIR/sglang_command.txt"
"${SGLANG_CMD[@]}" > "$SERVER_LOG" 2>&1 &
SERVER_PID=$!

capture_cache_metrics() {
{
echo "=== SGLang cache metrics snapshot $(date --iso-8601=seconds) ==="
curl -fsS "http://localhost:$PORT/metrics" 2>/dev/null \
| grep -E '^(sglang:(cache_hit_rate|cached_tokens_total|prompt_tokens_total|hicache_host_used_tokens|hicache_host_total_tokens|token_usage|num_requests_running|num_requests_waiting))' \
|| true
echo "============================================================"
} >> "$SERVER_LOG"
}

wait_for_server_ready --port "$PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID"

capture_cache_metrics
trap capture_cache_metrics EXIT

if [ "${EVAL_ONLY:-false}" = "true" ]; then
run_eval --port "$PORT"
else
build_replay_cmd "$RESULT_DIR"
REPLAY_CMD+=" --server-metrics http://localhost:$PORT/metrics"
run_agentic_replay_and_write_outputs "$RESULT_DIR"
fi
17 changes: 17 additions & 0 deletions configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8602,6 +8602,23 @@ minimaxm3-fp8-h200-vllm-agentic:
- { tp: 8, ep: 8, kv-offloading: none, conc-list: [2, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 18, 20] }
- { tp: 8, ep: 8, kv-offloading: dram, kv-offload-backend: { name: mooncake, version: "0.3.11.post1" }, conc-list: [5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 16, 18, 20] }

qwen3.5-fp4-b200-sglang-agentic-mtp:
image: lmsysorg/sglang:v0.5.16-cu130
model: nvidia/Qwen3.5-397B-A17B-NVFP4
model-prefix: qwen3.5
runner: cluster:b200-dgxc
precision: fp4
framework: sglang
multinode: false
scenarios:
agentic-coding:
- dram-utilization: 0.80
search-space:
- { tp: 4, ep: 1, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 4, 8, 12, 16, 20, 24, 28, 32, 40, 48, 56, 60, 62, 64] }
- { tp: 4, ep: 1, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [64, 66, 68, 70, 72, 76] }
- { tp: 2, ep: 1, spec-decoding: mtp, kv-offloading: none, conc-list: [1, 4, 8, 12, 14] }
- { tp: 2, ep: 1, spec-decoding: mtp, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [16, 18, 20, 22, 24, 28, 32] }

dsv4-fp4-b200-sglang-agentic-hicache:
image: lmsysorg/sglang:nightly-dev-cu13-20260707-b4155233
model: deepseek-ai/DeepSeek-V4-Pro
Expand Down
10 changes: 10 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -5364,3 +5364,13 @@
- "Use the shared AIPerf watchdog to cap whole-trajectory runtime idle gaps at 300 seconds"
- "Image: lmsysorg/sglang:v0.5.16-cu130"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2421

- config-keys:
- qwen3.5-fp4-b200-sglang-agentic-mtp
description:
- "Add Qwen3.5-397B-A17B NVFP4 AgentX benchmark on B200 with SGLang native NEXTN MTP"
- "Use the 256k trace dataset and golden synthetic acceptance length 3.39"
- "Use the shared AIPerf watchdog to cap whole-trajectory runtime idle gaps at 300 seconds"
- "Image: lmsysorg/sglang:v0.5.16-cu130"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2420

9 changes: 5 additions & 4 deletions runners/launch_b200-dgxc.sh
Original file line number Diff line number Diff line change
Expand Up @@ -451,10 +451,6 @@ EOF
else

SQUASH_FILE="/home/sa-shared/containers/$(echo "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh"
# Point the bench script at the local MODEL_PATH resolved above instead of
# pulling from the HF hub cache. Bench scripts skip `hf download` when
# MODEL is a local path.
export MODEL="$MODEL_PATH"
FRAMEWORK_SUFFIX=$([[ "$FRAMEWORK" == "trt" ]] && printf '_trt' || printf '')
SPEC_SUFFIX=$([[ "$SPEC_DECODING" == "mtp" ]] && printf '_mtp' || printf '')
# Prefer a framework-tagged script (e.g. dsv4_fp4_b200_vllm.sh) so models
Expand Down Expand Up @@ -488,6 +484,11 @@ else
salloc --partition=$SLURM_PARTITION --account=$SLURM_ACCOUNT --gres=gpu:$GPU_COUNT --exclusive --mem=0 --time="$SALLOC_TIME_LIMIT" --no-shell --job-name="$RUNNER_NAME"
JOB_ID=$(squeue --name="$RUNNER_NAME" -u "$USER" -h -o %A | head -n1)

# Point the bench script at the resolved MODEL_PATH instead of
# pulling from the HF hub cache. Bench scripts skip `hf download` when
# MODEL is a local path.
export MODEL="$MODEL_PATH"

# Use flock to serialize concurrent imports to the same squash file
# Override ENROOT_CACHE_PATH to avoid permission issues with system-wide cache on worker nodes
srun --jobid=$JOB_ID bash -c "
Expand Down