diff --git a/benchmarks/single_node/agentic/dsv41flash_fp4_mi355x_vllm_mtp.sh b/benchmarks/single_node/agentic/dsv41flash_fp4_mi355x_vllm_mtp.sh index 40ec8d58a..fa2ded18a 100755 --- a/benchmarks/single_node/agentic/dsv41flash_fp4_mi355x_vllm_mtp.sh +++ b/benchmarks/single_node/agentic/dsv41flash_fp4_mi355x_vllm_mtp.sh @@ -3,7 +3,8 @@ set -eo pipefail # DeepSeek-V4.1-Flash on MI355X: native DSpark and GPU-resident KV. # Follow upstream AMD defaults for Engram; storage behavior needs verification. -# Image: vllm/vllm-openai-rocm:deepseekv41-flash-0909 (configured in amd-master.yaml); GPU validation is pending. +# Image: vllm/vllm-openai-rocm:nightly-eed1f3d0c6043bd494424a22443ee198dd56f657 +# (configured in amd-master.yaml); GPU validation is pending. # https://github.com/vllm-project/recipes/blob/main/models/deepseek-ai/DeepSeek-V4.1-Flash.yaml source "$(dirname "$0")/../../benchmark_lib.sh" check_env_vars MODEL TP CONC KV_OFFLOADING TOTAL_CPU_DRAM_GB RESULT_DIR DURATION @@ -23,6 +24,14 @@ if [[ -n "${ROCR_VISIBLE_DEVICES:-}" ]]; then fi export VLLM_ROCM_USE_AITER=1 export VLLM_ROCM_USE_AITER_MOE=1 +# AITER's Triton MoE GEMM warns on every call that Gluon is unavailable and it +# is falling back to Triton. Gluon supports only gfx1250, so on gfx950 that is +# a fixed property rather than a condition worth reporting, and it was 98% of +# the lines in a gsm8k server log (411k of 417k, 30 MiB of 32 MiB). Every +# warning aiter.ops.triton emits is about Gluon availability, so raising the +# threshold loses nothing actionable here. Log hygiene only: the emits cost +# 0.03% of wall time per worker, so this is not a throughput change. +export AITER_TRITON_LOG_LEVEL=ERROR # DeepseekV41ForCausalLM is not torch-compiled upstream, so the default # cudagraph_mode=FULL_AND_PIECEWISE aborts at engine init with "piecewise CUDA # graphs unavailable" (run 34566727564). The model is built for the breakable @@ -39,8 +48,11 @@ export VLLM_ENGINE_READY_TIMEOUT_S=3600 export VLLM_USE_RUST_FRONTEND=1 export PYTHONUNBUFFERED=1 -# Match the sibling's scheduler headroom for AgentX subagent fan-out. -MAX_NUM_SEQS=$((2 * CONC)) +# Upstream default. The previous 2*CONC cap sat below AgentX's subagent +# fan-out, so at CONC=1 the engine admitted 2 requests and left the rest +# queued on scheduling capacity. Pinned rather than inherited so CAPTURE_SIZE +# below stays consistent with it. +MAX_NUM_SEQS=128 NUM_SPEC_TOKENS=5 CAPTURE_SIZE=1 while (( CAPTURE_SIZE < MAX_NUM_SEQS * (1 + NUM_SPEC_TOKENS) && CAPTURE_SIZE < 2048 )); do diff --git a/configs/amd-master.yaml b/configs/amd-master.yaml index 3d9a3dcca..2e061d32f 100644 --- a/configs/amd-master.yaml +++ b/configs/amd-master.yaml @@ -1753,9 +1753,11 @@ dsv4-fp4-mi355x-sglang-agentic-mtp: - { tp: 8, ep: 1, dp-attn: false, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [32, 48], spec-decoding: mtp } - { tp: 8, ep: 1, dp-attn: true, kv-offloading: dram, kv-offload-backend: { name: hicache }, conc-list: [128, 256], spec-decoding: mtp } -# Official upstream ROCm image from the vLLM recipe; MI355X runtime validation is pending. +# Upstream ROCm nightly rather than the deepseekv41-flash-0909 release tag: that +# tag predates vllm-project/vllm#56503, which moves the mHC delayed pre block off +# the eager Torch reference and onto AITER. MI355X runtime validation is pending. dsv41flash-fp4-mi355x-vllm-agentic-dspark: - image: vllm/vllm-openai-rocm:deepseekv41-flash-0909 + image: vllm/vllm-openai-rocm:nightly-eed1f3d0c6043bd494424a22443ee198dd56f657 model: deepseek-ai/DeepSeek-V4.1-Flash model-prefix: dsv41flash runner: cluster:mi355x-amds diff --git a/docs/configuration-procedures.md b/docs/configuration-procedures.md index d853f6493..98e265a58 100644 --- a/docs/configuration-procedures.md +++ b/docs/configuration-procedures.md @@ -403,4 +403,4 @@ The draft `dsv41flash-fp4-mi355x-vllm-agentic-dspark` recipe extends [#2958](htt Follow the AMD overrides in [upstream recipe #946](https://github.com/vllm-project/recipes/pull/946): `VLLM_ROCM_USE_AITER=1`, `VLLM_ROCM_USE_AITER_MOE=1`, and `--moe-backend aiter_triton_mxfp4_bf16`. The recipe pins `semianalysis_cc_traces_weka_062126` (the unfiltered corpus) via `WEKA_LOADER_OVERRIDE`. KV stays GPU-resident; Engram follows upstream AMD defaults. Do not copy the NVIDIA `--engram-config` option: upstream currently rejects it on ROCm. The MI355X launcher uses the shared HF cache and mounts this model's repository at `/ix`, and exports `INFMAX_CONTAINER_WORKSPACE=/ix` so AgentX dependencies and outputs resolve inside that mount. -**GPU validation is pending:** The recipe uses `vllm/vllm-openai-rocm:deepseekv41-flash-0909`, the image referenced by the [upstream vLLM recipe](https://recipes.vllm.ai/deepseek-ai/DeepSeek-V4.1-Flash?hardware=mi355x). The tag was published to Docker Hub on 2026-09-11 after AMD verification; it returned HTTP 404 in run 34466680355 before publication. Engram behavior and serving compatibility still require GPU validation. Follow the [AgentX procedure](./eval-agentx-procedures.md#7-run-agentx-fast-feedback-versus-canonical-evidence) for runtime evidence; local generation and registry metadata are not GPU proof. +**GPU validation is pending:** The recipe uses `vllm/vllm-openai-rocm:nightly-eed1f3d0c6043bd494424a22443ee198dd56f657` (digest `sha256:960228cf…`, published 2026-09-12). The `deepseekv41-flash-0909` tag referenced by the [upstream vLLM recipe](https://recipes.vllm.ai/deepseek-ai/DeepSeek-V4.1-Flash?hardware=mi355x) predates [vllm-project/vllm#56503](https://github.com/vllm-project/vllm/pull/56503), which moves the mHC delayed pre block off the eager Torch reference and onto AITER; that block is 85% of the decoder's kernel launches on this model, so the release tag leaves most of the decode cost on the table. The upstream recipe is being moved to the same nightly in [vllm-project/recipes#962](https://github.com/vllm-project/recipes/pull/962). Engram behavior and serving compatibility still require GPU validation. Follow the [AgentX procedure](./eval-agentx-procedures.md#7-run-agentx-fast-feedback-versus-canonical-evidence) for runtime evidence; local generation and registry metadata are not GPU proof. diff --git a/docs/configuration-procedures_zh.md b/docs/configuration-procedures_zh.md index 883f1f8e2..8046b94fc 100644 --- a/docs/configuration-procedures_zh.md +++ b/docs/configuration-procedures_zh.md @@ -381,4 +381,4 @@ python -m pytest utils/matrix_logic/ -v 遵循[上游配方 #946](https://github.com/vllm-project/recipes/pull/946) 的 AMD 设置:`VLLM_ROCM_USE_AITER=1`、`VLLM_ROCM_USE_AITER_MOE=1` 和 `--moe-backend aiter_triton_mxfp4_bf16`。配方通过 `WEKA_LOADER_OVERRIDE` 固定使用完整语料 `semianalysis_cc_traces_weka_062126`。KV 驻留 GPU;Engram 沿用上游 AMD 默认设置。不要复制 NVIDIA 的 `--engram-config` 选项:上游目前在 ROCm 上拒绝该选项。MI355X launcher 使用共享 HF 缓存,并将此模型的仓库挂载至 `/ix`,同时导出 `INFMAX_CONTAINER_WORKSPACE=/ix`,确保 AgentX 依赖与输出路径位于该挂载中。 -**GPU 验证尚待完成:** 配方使用 `vllm/vllm-openai-rocm:deepseekv41-flash-0909`,即[上游 vLLM 配方](https://recipes.vllm.ai/deepseek-ai/DeepSeek-V4.1-Flash?hardware=mi355x)引用的镜像。该标签经 AMD 验证后于 2026-09-11 发布到 Docker Hub;发布前在运行 34466680355 中返回 HTTP 404。Engram 行为与服务兼容性仍需 GPU 验证。请按 [AgentX 流程](./eval-agentx-procedures_zh.md) 获取运行时证据;本地矩阵生成和镜像元数据不等于 GPU 验证。 +**GPU 验证尚待完成:** 配方使用 `vllm/vllm-openai-rocm:nightly-eed1f3d0c6043bd494424a22443ee198dd56f657`(摘要 `sha256:960228cf…`,发布于 2026-09-12)。[上游 vLLM 配方](https://recipes.vllm.ai/deepseek-ai/DeepSeek-V4.1-Flash?hardware=mi355x)引用的 `deepseekv41-flash-0909` 标签早于 [vllm-project/vllm#56503](https://github.com/vllm-project/vllm/pull/56503),该 PR 将 mHC delayed pre 块从 eager Torch 参考实现切换到 AITER;该块占此模型解码器内核启动数的 85%,因此发布标签会浪费大部分解码开销。上游配方正在 [vllm-project/recipes#962](https://github.com/vllm-project/recipes/pull/962) 中切换到同一 nightly。Engram 行为与服务兼容性仍需 GPU 验证。请按 [AgentX 流程](./eval-agentx-procedures_zh.md) 获取运行时证据;本地矩阵生成和镜像元数据不等于 GPU 验证。 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 88c163461..d3539d98c 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -7399,3 +7399,17 @@ - "Disable adaptive verification in the EVAL_ONLY DSpark config as well. It trims verification requests on device, which the ROCm DeepseekV4IndexerBackend does not support, so the eval-only engine refused to start (run 34651830283, c32). Evals keep real block rejection; throughput settings are unchanged." - "EVAL_ONLY 的 DSpark 配置同样关闭自适应验证:它会在设备端裁剪验证请求,而 ROCm 的 DeepseekV4IndexerBackend 不支持该操作,导致仅评测引擎拒绝启动(运行 34651830283,c32)。评测仍保留真实块拒绝采样;吞吐设置不变。" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2962 + +- config-keys: + - dsv41flash-fp4-mi355x-vllm-agentic-dspark + description: + - "Switch to the upstream ROCm nightly image nightly-eed1f3d0c6043bd494424a22443ee198dd56f657. The deepseekv41-flash-0909 tag predates vllm-project/vllm#56503, which moves the mHC delayed pre block off the eager Torch reference onto AITER; that block is 85% of the decoder's kernel launches, measured 15.9-20.9x faster under cudagraph with gsm8k unchanged at 0.9719. The upstream recipe moves to the same nightly in vllm-project/recipes#962." + - "切换到上游 ROCm nightly 镜像 nightly-eed1f3d0c6043bd494424a22443ee198dd56f657。deepseekv41-flash-0909 标签早于 vllm-project/vllm#56503,该 PR 将 mHC delayed pre 块从 eager Torch 参考实现切换到 AITER;该块占解码器内核启动数的 85%,在 cudagraph 下实测加速 15.9-20.9 倍,gsm8k 保持 0.9719 不变。上游配方在 vllm-project/recipes#962 中切换到同一 nightly。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3035 + +- config-keys: + - dsv41flash-fp4-mi355x-vllm-agentic-dspark + description: + - "Raise max_num_seqs to the upstream default of 128. The 2*CONC cap sat below AgentX's subagent fan-out, so requests queued on scheduling capacity; vLLM reported queue times up to 16.2 s at CONC=1, and removing the cap took them to zero." + - "将 max_num_seqs 提升到上游默认值 128。此前的 2*CONC 上限低于 AgentX 的子代理扇出,导致请求因调度容量排队;vLLM 在 CONC=1 时报告排队时间最高达 16.2 秒,取消该上限后归零。" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3035