Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
18 commits
Select commit Hold shift + click to select a range
d5cd452
[AMD] [WIP] [AGENTX] Adding GLM5.2 MI355X Support - Agentx
giovanniguastiamd Aug 4, 2026
b7e7b89
[AMD] [WIP] [AGENTX] Adding GLM5.2 MI355X Support - Agentx
giovanniguastiamd Aug 4, 2026
ab95e49
[AMD] [WIP] [AGENTX] Adding GLM5.2 MI355X Support - Agentx
giovanniguastiamd Aug 4, 2026
8bf03a2
chore: update pr-link for glm5.2-fp4-mi355x-sglang-agentic-mtp
giovanniguastiamd Aug 4, 2026
a11fcf2
[AMD] [WIP] [AGENTX] GLM 5.2 - Update glm5.2_fp4_mi355x_sglang_mtp.sh
ajith-sirra-amd Aug 5, 2026
beb8cf6
Merge branch 'main' into amd/agentx-glm-5.2-sglang
ajith-sirra-amd Aug 5, 2026
6fa806b
[AMD] [WIP] [AGENTX] GLM 5.2 - Update perf-changelog.yaml
ajith-sirra-amd Aug 5, 2026
194f975
[AMD] [WIP] [AGENTX] GLM 5.2 - Update glm5.2_fp4_mi355x_sglang_mtp.sh
ajith-sirra-amd Aug 5, 2026
0175c27
fix(glm5.2-fp4-mi355x-sglang-mtp): remove trailing backslashes from S…
giovanniguastiamd Aug 5, 2026
6d65f91
fix(glm5.2-fp4-mi355x-sglang-mtp): increase watchdog-timeout to 3600s
giovanniguastiamd Aug 5, 2026
95dd745
[AMD] [WIP] [AGENTX] GLM 5.2 - Update glm5.2_fp4_mi355x_sglang_mtp.sh
ajith-sirra-amd Aug 5, 2026
6eca9be
fix(glm5.2-fp4-mi355x-sglang-agentic-mtp): set mem-fraction-static 0.…
giovanniguastiamd Aug 5, 2026
ae1943a
fix(glm5.2-fp4-mi355x-sglang-mtp): rebase onto main, re-append change…
giovanniguastiamd Aug 10, 2026
fd6b23d
Merge branch 'main' into amd/agentx-glm-5.2-sglang
seungrokj Aug 11, 2026
02c59f7
[AMD][AgentX] glm5.2 fp4 mi355x sglang mtp: set SGLANG_SIMULATE_ACC_L…
seungrokj Aug 11, 2026
fff9004
[AMD] [AGENTX] GLM 5.2 - Update glm5.2_fp4_mi355x_sglang_mtp.sh
ajith-sirra-amd Aug 11, 2026
532d328
[AMD] [AGENTX] GLM 5.2 - Update amd-master.yaml
ajith-sirra-amd Aug 11, 2026
d204ac0
[AMD] [AGENTX] GLM 5.2 Update Search Space
ajith-sirra-amd Aug 11, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
272 changes: 272 additions & 0 deletions benchmarks/single_node/agentic/glm5.2_fp4_mi355x_sglang_mtp.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,272 @@
#!/usr/bin/env bash
set -eo pipefail
set -x

source "$(dirname "$0")/../../benchmark_lib.sh"

export EVAL_FRAMEWORK="lm-eval"

check_env_vars MODEL TP CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION EP_SIZE DP_ATTENTION

if [[ -n "$SLURM_JOB_ID" ]]; then
echo "JOB $SLURM_JOB_ID running on $SLURMD_NODENAME"
fi

# ROCR/HIP visibility under slurm cgroups.
if [ -n "$ROCR_VISIBLE_DEVICES" ]; then
export HIP_VISIBLE_DEVICES="$ROCR_VISIBLE_DEVICES"
fi


if [[ -n "$MODEL_PATH" ]]; then
if [[ ! -d "$MODEL_PATH" || -z "$(ls -A "$MODEL_PATH" 2>/dev/null)" ]]; then
hf download "$MODEL" --local-dir "$MODEL_PATH"
fi
else
hf download "$MODEL"
export MODEL_PATH="$MODEL"
fi
rocm-smi || true
amd-smi || true

# A server killed on this node minutes earlier (previous job, crashed run)
# can still be draining its ~1.4 TB of HBM: KFD reclaim takes minutes, and
# booting into a half-drained node fails RCCL init with HIP 'unhandled cuda
# error' / 'invalid argument' (observed as the mooncake-c64 CI failure).
# Wait for the GPUs to come back before launching.
# Per-GPU threshold: idle nodes hold a small driver/firmware VRAM baseline
# (observed up to ~4%/GPU, node-dependent), while a draining or occupied
# GPU sits at 50-90%. Require every GPU <= 10%.
GPU_CLEAN=false
for i in $(seq 1 90); do
VRAM_MAX=$(rocm-smi --showmemuse 2>/dev/null | grep -oE "GPU Memory Allocated \(VRAM%\): [0-9]+" | awk '{if ($NF > m) m = $NF} END {print m+0}')
if [ "${VRAM_MAX:-0}" -le 10 ]; then echo "GPUs clean (vram%max=$VRAM_MAX after $((i*10))s)"; GPU_CLEAN=true; break; fi
echo "waiting for prior-job GPU memory reclaim: vram%max=$VRAM_MAX"; sleep 10
done
[ "$GPU_CLEAN" = "true" ] || { echo "Error: GPUs still draining prior job's memory after 15min" >&2; exit 1; }

resolve_trace_source
install_agentic_deps

SERVER_LOG="$RESULT_DIR/server.log"
ROUTER_LOG="$RESULT_DIR/router.log"
mkdir -p "$RESULT_DIR"

export PYTHONNOUSERSITE=1
# Agentic warmup dispatches hundreds of large prompts at once; allow up to
# 15 minutes of TCP progress before AIPerf declares a connection dead.
export AIPERF_HTTP_TCP_USER_TIMEOUT=900000
# AIPerf pins one pooled keep-alive connection per session (client-side
# keep-alive 300s) while uvicorn's default SGLANG_TIMEOUT_KEEP_ALIVE is 5s;
# inter-turn idle gaps can reuse a socket exactly as the server closes it.
# Outlast the client pool so the race cannot occur.
export SGLANG_TIMEOUT_KEEP_ALIVE=900
# The DSA indexer's top-k v2 kernel (default since v0.5.14) is JIT-compiled
# from CUDA-only source (cooperative_groups.h) and cannot build for gfx950;
# v1 dispatches to the precompiled HIP op in sgl-kernel (upstream MI355X CI
# runs DSA models the same way).
export SGLANG_OPT_USE_TOPK_V2=false

# HiCache L2 (host DRAM), optionally extended with Mooncake L3.
# KV_OFFLOADING=dram requires KV_OFFLOAD_BACKEND=hicache or mooncake.
#
# Per-arm L2 ratio (sizing rationale below) applies to both backends unless
# overridden via HICACHE_RATIO. TP arm (182.7 GB/rank device pool): the
# working set oversubscribes the device pool ~3x at conc 32, so the host
# tier is what carries the radix hits - ratio 1.5 (~2.9 TB pinned incl.
# sidecars) validates through the conc-24 long-context storm for the
# mooncake arm. The DP-attention arm (159.4 GB/rank) only runs at conc >=
# 32, where each DP rank's ~8 sessions nearly fit in its own device pool
# (~1.5-1.6M of 1.7M tokens at conc 64) and the host tier just absorbs
# overflow - ratio 1.5 boots but the host OOM killer takes the server
# mid-storm at conc 48, so it runs ratio 0.5 (~1.2 TB pinned, ~1.8 TB of
# load headroom) at negligible hit-rate cost. The hicache-only arm has no
# L3 to fall back on, so these ratios are unvalidated there - override with
# HICACHE_RATIO if the host OOMs or hit-rate is poor.
CACHE_ARGS=()
if agentic_kv_offload_enabled; then
if [ "$DP_ATTENTION" = "true" ]; then
HICACHE_RATIO="${HICACHE_RATIO:-0.5}"
else
HICACHE_RATIO="${HICACHE_RATIO:-1.5}"
fi
HICACHE_WRITE_POLICY="${HICACHE_WRITE_POLICY:-write_through}"
HICACHE_IO_BACKEND="${HICACHE_IO_BACKEND:-direct}"
HICACHE_MEM_LAYOUT="${HICACHE_MEM_LAYOUT:-page_first_direct}"
case "$KV_OFFLOAD_BACKEND" in
hicache)
echo "HiCache (GPU+host DRAM only): ratio=$HICACHE_RATIO, write_policy=$HICACHE_WRITE_POLICY, io_backend=$HICACHE_IO_BACKEND, mem_layout=$HICACHE_MEM_LAYOUT"
CACHE_ARGS=(
--enable-hierarchical-cache
--hicache-ratio "$HICACHE_RATIO"
--hicache-write-policy "$HICACHE_WRITE_POLICY"
--hicache-io-backend "$HICACHE_IO_BACKEND"
--hicache-mem-layout "$HICACHE_MEM_LAYOUT"
)
;;
mooncake)
L3_PER_RANK_GB="${L3_PER_RANK_GB:-40}"
python3 -c "from mooncake.store import MooncakeDistributedStore" >/dev/null
MOONCAKE_MASTER_PORT=$((PORT + 12000))
MOONCAKE_MASTER_LOG="$RESULT_DIR/mooncake_master.log"
MOONCAKE_CONFIG_PATH="$RESULT_DIR/mooncake_config.json"
cat > "$MOONCAKE_CONFIG_PATH" <<EOF
{
"local_hostname": "127.0.0.1",
"metadata_server": "P2PHANDSHAKE",
"master_server_address": "127.0.0.1:$MOONCAKE_MASTER_PORT",
"global_segment_size": "${L3_PER_RANK_GB}gb",
"local_buffer_size": "4gb",
"protocol": "tcp",
"device_name": ""
}
EOF
export SGLANG_HICACHE_MOONCAKE_CONFIG_PATH="$MOONCAKE_CONFIG_PATH"
mooncake_master --port "$MOONCAKE_MASTER_PORT" \
--default_kv_lease_ttl=120s \
--eviction_high_watermark_ratio=0.80 \
--eviction_ratio=0.10 > "$MOONCAKE_MASTER_LOG" 2>&1 &
MOONCAKE_MASTER_PID=$!
sleep 2
kill -0 "$MOONCAKE_MASTER_PID"
echo "HiCache+Mooncake: ratio=$HICACHE_RATIO, l3_per_rank=${L3_PER_RANK_GB} GB, dram_budget=${TOTAL_CPU_DRAM_GB} GB"
CACHE_ARGS=(
--enable-hierarchical-cache
--hicache-ratio "$HICACHE_RATIO"
--hicache-size 0
--hicache-write-policy "$HICACHE_WRITE_POLICY"
--hicache-io-backend "$HICACHE_IO_BACKEND"
--hicache-mem-layout "$HICACHE_MEM_LAYOUT"
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
)
;;
*)
echo "Error: unsupported KV_OFFLOAD_BACKEND '$KV_OFFLOAD_BACKEND' (expected: hicache or mooncake)" >&2
exit 1
;;
esac
fi

# Arm selection. TP arm keeps the FP8 sibling's cookbook batch-shaping
# bands.
#
# NOTE: the DP-attention path below is currently DORMANT (no dp-attn arms
# in amd-master.yaml): DSA + dp-attention hangs a collective under
# long-context prefill on ROCm v0.5.14 (watchdog kills the scheduler with
# zero completions; reproduced with and without HiCache, with and without
# the DSv4 DP collective envs; short prompts are fine). Re-enable the
# config arm once upstream fixes the DSA DP prefill path.
#
# When active, the DP-attention (DEP) arm fronts the DP ranks with sglang-router
# using consistent hashing on the AIPerf correlation id so multi-turn
# sessions stay on the DP rank holding their radix/hicache prefix, and
# widens chunked-prefill (whole-engine, /dp ranks) like the B300 sibling.
USE_SGLANG_ROUTER=false
SGLANG_BACKEND_PORT="$PORT"
PARALLEL_ARGS=(--tp "$TP" --ep-size "$EP_SIZE")
MEM_FRACTION_STATIC=0.85
if [ "$DP_ATTENTION" = "true" ]; then
USE_SGLANG_ROUTER=true
export AIPERF_HTTP_X_SMG_ROUTING_KEY_FROM_CORRELATION_ID=true
SGLANG_BACKEND_PORT=$((PORT + 1))
SGLANG_ROUTER_METRICS_PORT=$((PORT + 10000))
SGLANG_ROUTER_CMD=(python3 -m sglang_router.launch_router)
PARALLEL_ARGS+=(--dp "$TP" --enable-dp-attention)
CHUNKED_PREFILL_SIZE=32768
export AGENTIC_WARMUP_GRACE_PERIOD=3600
# Swap the DP gather collectives to gatherv/reduce-scatter on ROCm
# (dsv4_fp4_mi355x_sglang.sh precedent - the only green DP-attention
# config on this cluster/image): with the defaults the DSA DP path
# hangs a collective under long-context prefill load until the
# watchdog kills the scheduler (0/96 storm completions, twice).
export SGLANG_DP_USE_GATHERV=1
export SGLANG_DP_USE_REDUCE_SCATTER=1
export GPU_MAX_HW_QUEUES=5
elif [ "$CONC" -le 16 ]; then
# A full 131072-token prefill chunk needs ~7 GiB/rank of activation
# headroom on top of the static pool; pair it with mem-fraction 0.80
# like the FP8 sibling's low-conc band (0.85 OOMs the device mid-replay:
# "Tried to allocate 6.86 GiB ... 5.15 GiB is free", run 29751563205).
CHUNKED_PREFILL_SIZE=131072
MEM_FRACTION_STATIC=0.80
else
CHUNKED_PREFILL_SIZE=32768
export AGENTIC_WARMUP_GRACE_PERIOD=3600
fi
MAX_RUNNING_REQUESTS=$((1 * CONC))
[ "$MAX_RUNNING_REQUESTS" -gt 256 ] && MAX_RUNNING_REQUESTS=256
CUDA_GRAPH_MAX_BS=$MAX_RUNNING_REQUESTS

if [ "${EVAL_ONLY:-false}" != "true" ]; then
export SGLANG_SIMULATE_ACC_LEN=2.99
export SGLANG_SIMULATE_ACC_METHOD=match-expected
export SGLANG_SIMULATE_ACC_TOKEN_MODE=real-draft-token
fi

SGLANG_CMD=(
python3 -m sglang.launch_server
--model-path "$MODEL_PATH"
--served-model-name "$MODEL"
--host 0.0.0.0
--port "$SGLANG_BACKEND_PORT"
--trust-remote-code
"${PARALLEL_ARGS[@]}"
--kv-cache-dtype fp8_e4m3
--dsa-prefill-backend tilelang
--dsa-decode-backend tilelang
# GLM-5.2 emits the GLM-4.7-style tool-call format; glm47 is required for
# structured message.tool_calls (SWE-bench agentic evals die without it).
# The glm45 reasoning parser keeps hybrid thinking in reasoning_content.
--tool-call-parser glm47
--reasoning-parser glm45
--chunked-prefill-size "$CHUNKED_PREFILL_SIZE"
--mem-fraction-static "$MEM_FRACTION_STATIC"
--max-running-requests "$MAX_RUNNING_REQUESTS"
--cuda-graph-max-bs "$CUDA_GRAPH_MAX_BS"
--speculative-algorithm EAGLE
--speculative-num-steps 3
--speculative-eagle-topk 1
--speculative-num-draft-tokens 4
"${CACHE_ARGS[@]}"
--watchdog-timeout 1800
--enable-metrics
)

printf '%q ' "${SGLANG_CMD[@]}" | tee "$RESULT_DIR/sglang_command.txt"
printf '\n' | tee -a "$RESULT_DIR/sglang_command.txt"

echo "Starting SGLang server for MI355X..."
"${SGLANG_CMD[@]}" > "$SERVER_LOG" 2>&1 &
SERVER_PID=$!
echo "Server PID: $SERVER_PID"

wait_for_server_ready --port "$SGLANG_BACKEND_PORT" --server-log "$SERVER_LOG" --server-pid "$SERVER_PID"

if [ "$USE_SGLANG_ROUTER" = "true" ]; then
echo "Starting SGLang router on port $PORT for $TP DP ranks..."
"${SGLANG_ROUTER_CMD[@]}" \
--worker-urls "http://localhost:$SGLANG_BACKEND_PORT" \
--policy consistent_hashing \
--request-id-headers x-correlation-id \
--dp-aware \
--host 0.0.0.0 \
--port "$PORT" \
--prometheus-host 127.0.0.1 \
--prometheus-port "$SGLANG_ROUTER_METRICS_PORT" \
--connect-timeout-secs 900 \
--request-timeout-secs 14400 \
--disable-health-check \
--disable-retries > "$ROUTER_LOG" 2>&1 &
ROUTER_PID=$!
echo "Router PID: $ROUTER_PID"
wait_for_server_ready --port "$PORT" --server-log "$ROUTER_LOG" --server-pid "$ROUTER_PID"
fi

if [ "${EVAL_ONLY}" = "true" ]; then
run_eval --port "$PORT"
else
build_replay_cmd "$RESULT_DIR"
REPLAY_CMD+=" --server-metrics http://localhost:$SGLANG_BACKEND_PORT/metrics"
run_agentic_replay_and_write_outputs "$RESULT_DIR"
fi
19 changes: 19 additions & 0 deletions configs/amd-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -1615,3 +1615,22 @@ dsv4-fp8-mi325x-vllm-mtp:
# is 55.8%. conc128 passed cleanly on the memory-tightest SKU (MI300X);
# cap 8k1k MTP at 128 (normal holds 256, 1k1k holds 512).
- { tp: 8, conc-start: 4, conc-end: 128, spec-decoding: mtp }

# GLM-5.2 FP4 agentic-coding benchmark on MI355X via SGLang with MTP speculative
# decoding. TP=4 EP=4 with KV offloading to DRAM (hicache backend) to support
# long agentic context windows. Concurrency sweep [1, 2, 4, 8, 10].
glm5.2-fp4-mi355x-sglang-agentic-mtp:
image: lmsysorg/sglang-rocm:v0.5.16-rocm720-mi35x-20260728
model: amd/GLM-5.2-MXFP4
model-prefix: glm5.2
runner: cluster:mi355x-amds
precision: fp4
framework: sglang
multinode: false
scenarios:
agentic-coding:
- dram-utilization: 0.8
search-space:
- { tp: 4, ep: 4, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [1, 2, 4, 8, 10, 12, 16], spec-decoding: mtp }
- { tp: 8, ep: 8, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [1, 2, 4], spec-decoding: mtp }

7 changes: 7 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -5726,6 +5726,13 @@
- "Use supported header-based Dynamo session routing with the in-repo AIPerf build."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2520

- config-keys:
- glm5.2-fp4-mi355x-sglang-agentic-mtp
description:
- "Add GLM-5.2 FP4 SGLANG Single Node Agentic Support"
- "Image: lmsysorg/sglang-rocm:v0.5.16-rocm720-mi35x-20260728"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2488

- config-keys:
- dsv4-fp4-gb300-dynamo-sglang-agentic-agg
- dsv4-fp4-gb300-dynamo-sglang-agentic-disagg
Expand Down
Loading