From af0e8a124a8d94c90692f8ed59419716c52f0871 Mon Sep 17 00:00:00 2001 From: key4ng Date: Tue, 15 Sep 2026 21:43:06 -0700 Subject: [PATCH 1/2] ci(bench): move the BFCL and tau2 GLM legs from GLM-5.2-FP8 to GLM-5.3-Flash Replace the glm-5.2 leg (zai-org/GLM-5.2-FP8, ~744GB, whole node, arms sequential) with glm-5.3-flash (zai-org/GLM-5.3-Flash) in nightly-bfcl and nightly-tau2. - GLM-5.3-Flash is a 321B/18B-active native-FP8 multimodal MoE (~306GiB, KDA + NoPE sparse MLA) that fits a TP=4 half-node, so the leg becomes a concurrent TP=4 Blackwell leg like minimax-m2.7 / kimi-k2.6 and deepseek-v4.1 is now the only sequential whole-node leg. - Flags follow the vLLM recipe (models/zai-org/GLM-5.3-Flash.yaml): glm47/glm45 parsers, --kv-cache-dtype fp8 on Blackwell. SMG keeps glm47_moe/glm45 (GLM-5.3 retains the GLM-4.7 tool-call markers). - Support is on vLLM main only (vllm-project/vllm#53906), so the leg reuses the per-commit wheel override introduced for deepseek-v4.1; the pinned commit already contains Glm5Next. - Startup ceiling 3600s per the recipe's VLLM_ENGINE_READY_TIMEOUT_S; run timeouts back to the concurrent-leg norms. Update the bfcl/tau2 READMEs to match. Signed-off-by: key4ng --- .github/workflows/nightly-bfcl.yml | 37 +++++++++++++++++++----------- .github/workflows/nightly-tau2.yml | 31 ++++++++++++++----------- scripts/bfcl/README.md | 11 +++++---- scripts/tau2/README.md | 4 ++-- 4 files changed, 48 insertions(+), 35 deletions(-) diff --git a/.github/workflows/nightly-bfcl.yml b/.github/workflows/nightly-bfcl.yml index 0288bbc59..f62b014b7 100644 --- a/.github/workflows/nightly-bfcl.yml +++ b/.github/workflows/nightly-bfcl.yml @@ -23,7 +23,7 @@ on: workflow_dispatch: inputs: only: - description: "Run only this matrix leg (qwen3.8|gpt-oss|deepseek-v4.1|minimax-m2.7|kimi-k2.6|glm-5.2); empty = all" + description: "Run only this matrix leg (qwen3.8|gpt-oss|deepseek-v4.1|minimax-m2.7|kimi-k2.6|glm-5.3-flash); empty = all" required: false default: "" model: @@ -162,7 +162,7 @@ jobs: # Concurrent TP=4 per arm (0-3 + 4-7). # DeepSeek-V4.1-Flash is ~765GB on disk (552B backbone + 196B int8 Engram, # all GPU-resident), so it outgrows a TP=4 half-node and runs SEQUENTIAL - # on the whole node like glm-5.2. SMG: deepseek_v41 tool/reasoning parsers + # on the whole node (the only such leg). SMG: deepseek_v41 tool/reasoning parsers # + native renderer (#2524/#2525/#2527). vLLM: V4.1 is on main only # (vllm-project/vllm#56208 + #56228, 2026-09-10), not in the 0.27.1 CI # pin, so this leg installs a per-commit main wheel (vllm_commit below). @@ -208,20 +208,29 @@ jobs: "vllm_extra": "--trust-remote-code", "gpu_mem": "0.90", "startup_timeout": "2400", "run_timeout": "14400", "model_cache": "/raid/models", "arm_mode": "concurrent", "max_model_len": "auto"}, - # GLM-5.2-FP8 (~744GB) needs the whole 8-GPU node, so it runs SEQUENTIAL - # (arm A then arm B; gpu_b unused) unlike the concurrent half-node legs. - # vLLM recipe: TP=8, glm47/glm45, --kv-cache-dtype fp8_e4m3. SMG passes - # glm47_moe explicitly (org-prefixed served name won't auto-detect). - {"name": "glm-5.2", "runner": "blackwell", "tp": 8, - "gpu_a": "0,1,2,3,4,5,6,7", "gpu_b": "", # sequential: arm B reuses GPU_A (whole node) - "model": "zai-org/GLM-5.2-FP8", "bfcl_model": "zai-org/GLM-5.2-FP8-FC", + # GLM-5.3-Flash: 321B/18B-active native-FP8 multimodal MoE (~306GiB, KDA + + # NoPE sparse MLA). Fits a TP=4 half-node, so unlike GLM-5.2-FP8 (~744GB, + # whole node) it runs concurrent like the other Blackwell legs. vLLM recipe + # (vllm-project/recipes models/zai-org/GLM-5.3-Flash.yaml): glm47/glm45, + # --kv-cache-dtype fp8 on Blackwell, FlashInfer >=0.6.18. Support is on + # vLLM main only (vllm-project/vllm#53906, 2026-09-03), so it reuses the + # deepseek-v4.1 leg's per-commit main wheel. SMG passes glm47_moe + # explicitly (org-prefixed served name won't auto-detect). + {"name": "glm-5.3-flash", "runner": "blackwell", "tp": 4, + "gpu_a": "0,1,2,3", "gpu_b": "4,5,6,7", + "model": "zai-org/GLM-5.3-Flash", "bfcl_model": "zai-org/GLM-5.3-Flash-FC", "vllm_tool": "glm47", "vllm_reason": "glm45", "smg_tool": "glm47_moe", "smg_reason": "glm45", - # FP8 load + DeepGEMM warmup on the largest leg: generous startup ceiling. - "vllm_extra": "--trust-remote-code --kv-cache-dtype fp8_e4m3", "gpu_mem": "0.90", "startup_timeout": "3000", - # Sequential: both arms share the 360m job, healthy total 2h52m on 2026-08-05. + "vllm_extra": "--trust-remote-code --kv-cache-dtype fp8", + "vllm_commit": "9b959b86577c082c0b2bf9e2c22263255a36ad83", + "vllm_version": "0.28.1rc1.dev651+g9b959b865", + # Recipe sets VLLM_ENGINE_READY_TIMEOUT_S=3600: FP8 load + KDA/sparse-MLA + # JIT on a new arch. Kept at the V4.1 leg's ceiling. + "gpu_mem": "0.90", "startup_timeout": "3600", + # Per-arm cap carried over from the GLM-5.2 leg (healthy 2h52m for both + # arms serially); arms are concurrent here, so this is generous. "run_timeout": "7200", - "model_cache": "/raid/models", "arm_mode": "sequential", "max_model_len": "auto"}, + "model_cache": "/raid/models", "arm_mode": "concurrent", "max_model_len": "auto"}, ] only = os.environ.get("ONLY", "") valid_names = [l["name"] for l in legs] @@ -244,7 +253,7 @@ jobs: matrix: ${{ fromJSON(needs.setup.outputs.matrix) }} runs-on: ${{ matrix.runner }} # Generous: the nightly set adds multi_turn (state simulation), much slower - # than the single-turn AST/live categories. The sequential glm-5.2 leg needs + # than the single-turn AST/live categories. The sequential deepseek-v4.1 leg needs # the most headroom — two whole-node arms (startup + scoring) run serially, # not concurrently like the half-node legs. Ceiling only; fast legs finish early. timeout-minutes: 360 diff --git a/.github/workflows/nightly-tau2.yml b/.github/workflows/nightly-tau2.yml index 0b5ce631a..108c7a2b9 100644 --- a/.github/workflows/nightly-tau2.yml +++ b/.github/workflows/nightly-tau2.yml @@ -8,7 +8,7 @@ # the ONLY variable is the frontend (tokenization + tool/reasoning parsing). # # Runs nightly across retail, airline, and telecom domains over a 6-model matrix -# (qwen3.8 + gpt-oss on H100; deepseek-v4.1, minimax-m2.7, kimi-k2.6, glm-5.2 on +# (qwen3.8 + gpt-oss on H100; deepseek-v4.1, minimax-m2.7, kimi-k2.6, glm-5.3-flash on # Blackwell). The 4 Blackwell legs are num_tasks-capped to bound GPU time and # gpt-5.2 spend (tau2 is far heavier per-leg than bfcl: multi-turn + per-turn API # spend). On PRs that touch this pipeline ALL legs run on a tiny retail / 1-trial / @@ -31,7 +31,7 @@ on: workflow_dispatch: inputs: only: - description: "Run only this matrix leg (qwen3.8|gpt-oss|deepseek-v4.1|minimax-m2.7|kimi-k2.6|glm-5.2); empty = all" + description: "Run only this matrix leg (qwen3.8|gpt-oss|deepseek-v4.1|minimax-m2.7|kimi-k2.6|glm-5.3-flash); empty = all" required: false default: "" model: @@ -166,7 +166,7 @@ jobs: # Concurrent TP=4 per arm (0-3 + 4-7); num_tasks-capped. # DeepSeek-V4.1-Flash is ~765GB on disk (552B backbone + 196B int8 Engram, # all GPU-resident), so it outgrows a TP=4 half-node and runs SEQUENTIAL - # on the whole node like glm-5.2. SMG: deepseek_v41 tool/reasoning parsers + # on the whole node (the only such leg). SMG: deepseek_v41 tool/reasoning parsers # + native renderer (#2524/#2525/#2527). vLLM: V4.1 is on main only # (vllm-project/vllm#56208 + #56228, 2026-09-10), not in the 0.27.1 CI # pin, so this leg installs a per-commit main wheel (vllm_commit below). @@ -201,18 +201,21 @@ jobs: "vllm_extra": "--trust-remote-code", "gpu_mem": "0.90", "startup_timeout": "2400", "model_cache": "/raid/models", "arm_mode": "concurrent", "max_model_len": "auto", "num_tasks": 30, "run_timeout": "5400"}, - # GLM-5.2-FP8 (~744GB) needs the whole 8-GPU node, so it runs SEQUENTIAL - # (arm A then arm B; gpu_b unused) unlike the concurrent half-node legs. - {"name": "glm-5.2", "runner": "blackwell", "tp": 8, - "gpu_a": "0,1,2,3,4,5,6,7", "gpu_b": "", - "model": "zai-org/GLM-5.2-FP8", + # GLM-5.3-Flash (~306GiB FP8) fits a TP=4 half-node, so it runs concurrent, + # unlike the whole-node GLM-5.2-FP8 leg it replaces. Flags per the vLLM + # recipe; per-commit main wheel shared with deepseek-v4.1 (see nightly-bfcl.yml). + {"name": "glm-5.3-flash", "runner": "blackwell", "tp": 4, + "gpu_a": "0,1,2,3", "gpu_b": "4,5,6,7", + "model": "zai-org/GLM-5.3-Flash", "vllm_tool": "glm47", "vllm_reason": "glm45", "smg_tool": "glm47_moe", "smg_reason": "glm45", - "vllm_extra": "--trust-remote-code --kv-cache-dtype fp8_e4m3", - "gpu_mem": "0.90", "startup_timeout": "3000", "model_cache": "/raid/models", - "arm_mode": "sequential", "max_model_len": "auto", "num_tasks": 30, - # Sequential: both arms share the 360m job, so half the per-domain budget. - "run_timeout": "2700"}, + "vllm_extra": "--trust-remote-code --kv-cache-dtype fp8", + "vllm_commit": "9b959b86577c082c0b2bf9e2c22263255a36ad83", + "vllm_version": "0.28.1rc1.dev651+g9b959b865", + "gpu_mem": "0.90", "startup_timeout": "3600", "model_cache": "/raid/models", + "arm_mode": "concurrent", "max_model_len": "auto", "num_tasks": 30, + # Concurrent arms: full per-domain budget like the other half-node legs. + "run_timeout": "5400"}, ] only = os.environ.get("ONLY", "") valid = [l["name"] for l in legs] @@ -235,7 +238,7 @@ jobs: matrix: ${{ fromJSON(needs.setup.outputs.matrix) }} runs-on: ${{ matrix.runner }} # Generous ceiling: multi-turn state simulation is slow, and the sequential - # glm-5.2 leg scores two whole-node arms serially. Ceiling only; fast legs + # deepseek-v4.1 leg scores two whole-node arms serially. Ceiling only; fast legs # finish early. --max-concurrency keeps the full run within budget. timeout-minutes: 360 permissions: diff --git a/scripts/bfcl/README.md b/scripts/bfcl/README.md index c9524b45f..7e74db4f7 100644 --- a/scripts/bfcl/README.md +++ b/scripts/bfcl/README.md @@ -67,6 +67,7 @@ Key env knobs for `launch_arm.sh`: `BFCL_GPU` (CUDA_VISIBLE_DEVICES, e.g. `0,1`) | DeepSeek-V4.1-Flash (`deepseek-v4.1`) | `blackwell` | 8 (seq) | `deepseek_v41` / `deepseek_v41` (+`--tokenizer-mode deepseek_v41 --trust-remote-code`; per-commit vLLM main wheel) | `deepseek_v41` / `deepseek_v41` | | MiniMax-M2.7 (`minimax-m2.7`) | `blackwell` | 4 | `minimax_m2` / `minimax_m2` (+`--trust-remote-code`) | `minimax_m2` / `minimax` | | Kimi-K2.6 int4 (`kimi-k2.6`) | `blackwell` | 4 | `kimi_k2` / `kimi_k2` (+`--trust-remote-code`) | `kimik2` / `kimi_k25`† | +| GLM-5.3-Flash (`glm-5.3-flash`) | `blackwell` | 4 | `glm47` / `glm45` (+`--trust-remote-code --kv-cache-dtype fp8`; per-commit vLLM main wheel) | `glm47_moe` / `glm45` | > **gpt-oss has no SMG tool-call-parser.** SMG handles gpt-oss through its harmony > pipeline (`model_gateway/src/routers/grpc/harmony/`), auto-activated by @@ -89,8 +90,8 @@ The nightly (`.github/workflows/nightly-bfcl.yml`) runs the A/B as a GitHub Acti matrix — one leg per model, `fail-fast: false`, each on its own runner: - `4-gpu-h100` — Qwen3.8-27B and gpt-oss-120b, TP=2 per arm (GPUs 0,1 + 2,3). -- `blackwell` (B200) — MiniMax-M2.7 and Kimi-K2.6 int4, TP=4 per arm (GPUs 0-3 + 4-7); - DeepSeek-V4.1-Flash and GLM-5.2-FP8 need the whole node (TP=8, arms sequential). +- `blackwell` (B200) — MiniMax-M2.7, Kimi-K2.6 int4 and GLM-5.3-Flash, TP=4 per arm + (GPUs 0-3 + 4-7); DeepSeek-V4.1-Flash needs the whole node (TP=8, arms sequential). All legs use `max_model_len` **32768**: the `multi_turn` categories emit ~18k-token prompts that 400'd ("decoder prompt longer than the maximum model length") at 16384. @@ -101,7 +102,7 @@ Each leg sets `arm_mode`: - **concurrent** (the half-node legs) — both arms serve at once on opposite GPU halves; `run_ab.py` scores them **in parallel** (separate servers/GPUs, no contention) and diffs. -- **sequential** (`deepseek-v4.1`, `glm-5.2`) — for a model that needs the whole node (TP=8) +- **sequential** (`deepseek-v4.1`) — for a model that needs the whole node (TP=8) so the arms can't coexist: `run_ab.py --score-arm` scores arm A alone → tears it down → scores arm B alone → `--diff-baseline/--diff-candidate` compares the two saved score files. Flip a leg's `arm_mode` to enable it. @@ -118,8 +119,8 @@ is a tiny non-live subset (`simple_python,irrelevance`) for every leg. A leg whose model the pinned vLLM release (`scripts/ci_install_vllm.sh`) cannot serve sets `vllm_commit` + `vllm_version` in its matrix entry; the job then swaps in that per-commit main wheel from `wheels.vllm.ai/` for **both** arms (the A/B stays -engine-identical). `deepseek-v4.1` uses this until a vLLM release ships V4.1 and the -CI pin moves; drop the two keys then. +engine-identical). `deepseek-v4.1` and `glm-5.3-flash` share one such wheel until a +vLLM release ships both models and the CI pin moves; drop the keys then. ## Gotchas discovered while bringing this up (read before debugging) diff --git a/scripts/tau2/README.md b/scripts/tau2/README.md index b5cad31a6..a2d28e39a 100644 --- a/scripts/tau2/README.md +++ b/scripts/tau2/README.md @@ -87,7 +87,7 @@ where `` is the path you pass to the **required** `--data-dir` flag (t Mirrors `nightly-bfcl.yml`'s 6-leg matrix. The 2 H100 legs run full task sets; the 4 Blackwell legs are capped at `num_tasks=30`/domain (tau2 is multi-turn + spends -gpt-5.2 per turn). `glm-5.2` runs **sequential** (whole 8-GPU node per arm — `run_ab.py +gpt-5.2 per turn). `deepseek-v4.1` runs **sequential** (whole 8-GPU node per arm — `run_ab.py --score-arm` each arm, then `--diff`); the rest run both arms concurrently on opposite GPU halves. On PRs **all** legs run on a tiny retail / 1-trial / few-task subset — a quick "does each leg launch + parse + score" smoke (the heavy Blackwell legs are @@ -100,7 +100,7 @@ dominated by model-load time, serialized by a host lock, so a PR run is not fast | deepseek-v4.1 | deepseek-ai/DeepSeek-V4.1-Flash | blackwell (8, seq) | `deepseek_v41` / `deepseek_v41` | `deepseek_v41` / `deepseek_v41` | | minimax-m2.7 | MiniMaxAI/MiniMax-M2.7 | blackwell (4) | `minimax_m2` / `minimax_m2` | `minimax_m2` / `minimax` | | kimi-k2.6 | moonshotai/Kimi-K2.6 | blackwell (4) | `kimi_k2` / `kimi_k2` | `kimik2` / `kimi_k25` | -| glm-5.2 | zai-org/GLM-5.2-FP8 | blackwell (8, seq) | `glm47` / `glm45` | `glm47_moe` / `glm45` | +| glm-5.3-flash | zai-org/GLM-5.3-Flash | blackwell (4) | `glm47` / `glm45` | `glm47_moe` / `glm45` | > Dispatch `only=` runs a single leg; `model=` overrides its weights. SKU ids and > vLLM parser names may shift; confirm against the installed vLLM build: From ebabb3ba1b08ed857e9e33a437c6b9fc81b69c17 Mon Sep 17 00:00:00 2001 From: key4ng Date: Tue, 15 Sep 2026 22:15:45 -0700 Subject: [PATCH 2/2] ci(bench): re-pin the glm-5.3-flash vLLM wheel alongside deepseek-v4.1 Follow the deepseek-v4.1 re-pin to a31ec3a68bbbedd4b5d59490c762bcf32abf1a17 (0.29.1rc1.dev185+ga31ec3a68): the earlier commit shipped the DeepSeek-V4.1 classes without a registry entry. Glm5NextForConditionalGeneration is registered at the new commit as well, so both main-only legs keep sharing one wheel. Signed-off-by: key4ng --- .github/workflows/nightly-bfcl.yml | 4 ++-- .github/workflows/nightly-tau2.yml | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/.github/workflows/nightly-bfcl.yml b/.github/workflows/nightly-bfcl.yml index f62b014b7..ec9ae0426 100644 --- a/.github/workflows/nightly-bfcl.yml +++ b/.github/workflows/nightly-bfcl.yml @@ -222,8 +222,8 @@ jobs: "vllm_tool": "glm47", "vllm_reason": "glm45", "smg_tool": "glm47_moe", "smg_reason": "glm45", "vllm_extra": "--trust-remote-code --kv-cache-dtype fp8", - "vllm_commit": "9b959b86577c082c0b2bf9e2c22263255a36ad83", - "vllm_version": "0.28.1rc1.dev651+g9b959b865", + "vllm_commit": "a31ec3a68bbbedd4b5d59490c762bcf32abf1a17", + "vllm_version": "0.29.1rc1.dev185+ga31ec3a68", # Recipe sets VLLM_ENGINE_READY_TIMEOUT_S=3600: FP8 load + KDA/sparse-MLA # JIT on a new arch. Kept at the V4.1 leg's ceiling. "gpu_mem": "0.90", "startup_timeout": "3600", diff --git a/.github/workflows/nightly-tau2.yml b/.github/workflows/nightly-tau2.yml index 108c7a2b9..b200cd8fd 100644 --- a/.github/workflows/nightly-tau2.yml +++ b/.github/workflows/nightly-tau2.yml @@ -210,8 +210,8 @@ jobs: "vllm_tool": "glm47", "vllm_reason": "glm45", "smg_tool": "glm47_moe", "smg_reason": "glm45", "vllm_extra": "--trust-remote-code --kv-cache-dtype fp8", - "vllm_commit": "9b959b86577c082c0b2bf9e2c22263255a36ad83", - "vllm_version": "0.28.1rc1.dev651+g9b959b865", + "vllm_commit": "a31ec3a68bbbedd4b5d59490c762bcf32abf1a17", + "vllm_version": "0.29.1rc1.dev185+ga31ec3a68", "gpu_mem": "0.90", "startup_timeout": "3600", "model_cache": "/raid/models", "arm_mode": "concurrent", "max_model_len": "auto", "num_tasks": 30, # Concurrent arms: full per-domain budget like the other half-node legs.