From 0d8d0ffba3a09d03b93c0504d9fcef71a49c818d Mon Sep 17 00:00:00 2001 From: "Zhang, Jiejing" Date: Tue, 28 Jul 2026 03:33:35 +0000 Subject: [PATCH] fix(image): patch sglang for GLM-5.2 MTP nextn quark-exclude (backport #30265) The sglang v0.5.15.post1 base predates sgl-project/sglang#30265, so GLM-5.2 EAGLE/MTP crashes at draft weight-load: the MTP layer (index=num_hidden_layers) is entirely bf16, and its eh_proj is listed in the quark `exclude`, but sglang's DeepseekV3ForCausalLMNextN checks the BARE layer prefix `model.layers.` (not the submodule-level `....eh_proj` exclude entry), so should_ignore_layer() returns False, eh_proj is built as an MXFP4 uint8 param, and load asserts (param [6144,6144] uint8 vs bf16 [6144,12288]). Add a self-locating, idempotent Python patch (deploy/docker/patches/sglang/) that probes the eh_proj submodule instead of the bare layer -> nextn_quant_config=None -> whole bf16 MTP layer. Wire a patch loop into Dockerfile.sglang (mirrors the vLLM one). Verified: coherent GLM-5.2 MTP output on the v0.5.15.post1 base. Temporary: no-ops once the base sglang carries #30265 (merged upstream 2026-07-08); drop deploy/docker/patches/sglang/ + the loop when we bump the base sglang. Co-Authored-By: Claude Opus 4.8 (1M context) Claude-Session: https://claude.ai/code/session_01Pw7JSvdQb796xh5pEcctLz Signed-off-by: Zhang, Jiejing --- deploy/docker/Dockerfile.sglang | 10 ++++ .../sglang/patch_glm52_nextn_quark_exclude.py | 59 +++++++++++++++++++ 2 files changed, 69 insertions(+) create mode 100644 deploy/docker/patches/sglang/patch_glm52_nextn_quark_exclude.py diff --git a/deploy/docker/Dockerfile.sglang b/deploy/docker/Dockerfile.sglang index e85cb35fb..e64f6960a 100644 --- a/deploy/docker/Dockerfile.sglang +++ b/deploy/docker/Dockerfile.sglang @@ -42,6 +42,16 @@ COPY pyproject.toml README.md ./ COPY infera ./infera RUN pip install --no-cache-dir ".[sglang]" +# ---- sglang Python patches (GLM-5.2 MTP nextn quark-exclude; backport of sglang #30265) ---- +# The v0.5.15.post1 base predates sgl-project/sglang#30265, so GLM-5.2 EAGLE/MTP crashes at +# draft weight-load: the bf16 MTP layer's eh_proj is built as an MXFP4 param because sglang +# checks the bare layer prefix (not the submodule-level exclude entry). Each patch is +# self-locating, idempotent, and no-ops once the base sglang carries the fix — drop the patch +# dir then. Modeled on the vLLM patch loop. +COPY deploy/docker/patches/sglang/ /tmp/sglang-patches/ +RUN for f in /tmp/sglang-patches/*.py; do echo "[sglang-patch] $f"; python "$f" || echo "[sglang-patch] skipped $f"; done \ + && rm -rf /tmp/sglang-patches + # ---- Rust router (multi-core data plane; --router-backend rust) ---- # The ROCm base ships cargo; use it (install a minimal toolchain only if # absent). Needs cc for linking and libclang (onig_sys/bindgen). Removes diff --git a/deploy/docker/patches/sglang/patch_glm52_nextn_quark_exclude.py b/deploy/docker/patches/sglang/patch_glm52_nextn_quark_exclude.py new file mode 100644 index 000000000..d629fb740 --- /dev/null +++ b/deploy/docker/patches/sglang/patch_glm52_nextn_quark_exclude.py @@ -0,0 +1,59 @@ +#!/usr/bin/env python3 +# Copyright (c) 2026, Advanced Micro Devices, Inc. All rights reserved. +# SPDX-License-Identifier: MIT +"""Temporary backport of sgl-project/sglang#30265 — GLM-5.2 MTP (nextn) quark-exclude fix. + +WHAT: GLM-5.2's MTP layer (index = num_hidden_layers) is ENTIRELY bf16/unquantized — +the quark quantization_config lists `model.layers..eh_proj` (and the layer's other +submodules) in `exclude`. sglang's DeepseekV3ForCausalLMNextN disables nextn quant only +when the BARE layer prefix `model.layers.` is in exclude_layers, but the exclude +entries are submodule-level (`....eh_proj`), so `should_ignore_layer()` returns False, +`eh_proj` is built as an MXFP4 (uint8) param, and draft weight-load dies: + AssertionError: param.shape=[6144,6144] uint8 vs loaded_weight.shape=[6144,12288] bf16. + +FIX: probe the `eh_proj` submodule (an exact exclude entry) instead of the bare layer, so +the match succeeds -> nextn_quant_config=None -> the whole (bf16) MTP layer is built bf16. +Verified: coherent GLM-5.2 MTP output on the v0.5.15.post1 base. + +UPSTREAM: sgl-project/sglang#30265 (merged 2026-07-08) is the full fix (a dedicated +GlmMoeDsaForCausalLMNextN class). Our base image ships sglang v0.5.15.post1 which predates +it. DROP THIS PATCH once the base sglang carries #30265 (the anchor below will be gone and +this becomes a no-op). + +Self-locating, idempotent, no-op if the anchor is absent (sglang refactored / already fixed). +""" + +import importlib.util +import sys +from pathlib import Path + + +def _target(): + spec = importlib.util.find_spec("sglang") + if not spec or not spec.origin: + return None + f = Path(spec.origin).parent / "srt" / "models" / "deepseek_nextn.py" + return f if f.exists() else None + + +def main(): + f = _target() + if f is None: + print("[glm52-nextn] sglang deepseek_nextn.py not found — skipping") + return 0 + src = f.read_text() + old = 'ckpt_prefix = f"model.layers.{config.num_hidden_layers}"' + new = 'ckpt_prefix = f"model.layers.{config.num_hidden_layers}.eh_proj"' + if new in src: + print("[glm52-nextn] already patched — skipping") + return 0 + if old not in src: + print("[glm52-nextn] anchor absent (sglang has #30265 / refactored?) — skipping") + return 0 + f.write_text(src.replace(old, new, 1)) + print(f"[glm52-nextn] patched {f} (GLM-5.2 MTP eh_proj quark-exclude; backport of #30265)") + return 0 + + +if __name__ == "__main__": + sys.exit(main())