diff --git a/.gitattributes b/.gitattributes index dbd27938..a59958fc 100644 --- a/.gitattributes +++ b/.gitattributes @@ -1 +1,3 @@ patches/releases/**/integration.patch -whitespace +# Captured diagnostics retain their original bytes for evidence checksums. +recipes/glm53/evidence/**/*.log -whitespace diff --git a/recipes/glm53/Dockerfile.glm53-cache-contracts b/recipes/glm53/Dockerfile.glm53-cache-contracts new file mode 100644 index 00000000..4a5fe1e6 --- /dev/null +++ b/recipes/glm53/Dockerfile.glm53-cache-contracts @@ -0,0 +1,88 @@ +# syntax=docker/dockerfile:1.7 +ARG BASE_IMAGE=voipmonitor/vllm@sha256:93ac5228f1cbde2182ca294d8b479259144742af2756a49ff207dd245429bf43 +ARG FLASHINFER_WHEELS_IMAGE=voipmonitor/vllm@sha256:79edbc91874d9468e3e6268e1584503e3dec55f2a4d3bdd70d5c43e9b41675c7 +ARG LMCACHE_NATIVE_IMAGE=localinferencelab/vllm@sha256:3ae04d964f8e7e936ef4b34dd75406b9169fead8062fb8b4b1634f3bbf512827 +FROM ${FLASHINFER_WHEELS_IMAGE} AS flashinfer-wheels +FROM ${LMCACHE_NATIVE_IMAGE} AS lmcache-build +COPY --from=source_bundles /lmcache.bundle /source-bundles/lmcache.bundle +COPY --from=source_bundles /uv /source-bundles/uv +COPY install_source_bundle.sh /build-tools/install_source_bundle.sh +COPY pack_lmcache_native_reuse.py /build-tools/pack_lmcache_native_reuse.py +ARG LMCACHE_COMMIT +ARG LMCACHE_TREE +ARG LMCACHE_BUNDLE_SHA256 +ARG LMCACHE_VERSION +ARG LMCACHE_NATIVE_MODE=reuse-all +ARG UV_SHA256 +WORKDIR /tmp +RUN test "$(sha256sum /source-bundles/lmcache.bundle | cut -d' ' -f1)" = "${LMCACHE_BUNDLE_SHA256}" \ + && test "$(sha256sum /source-bundles/uv | cut -d' ' -f1)" = "${UV_SHA256}" \ + && bash /build-tools/install_source_bundle.sh /source-bundles/lmcache.bundle \ + "${LMCACHE_COMMIT}" "${LMCACHE_TREE}" /tmp/lmcache-source \ + && /source-bundles/uv pip install --python /opt/venv/bin/python \ + 'setuptools>=77.0.3,<81' setuptools_scm wheel packaging ninja \ + && cd /tmp/lmcache-source \ + && case "${LMCACHE_NATIVE_MODE}" in \ + reuse-all) export NO_NATIVE_EXT=1 ;; \ + cpu-rebuild-cuda-reuse) unset NO_NATIVE_EXT; export NO_GPU_EXT=1 MAX_JOBS=8 ;; \ + *) echo "Unsupported LMCache native mode" >&2; exit 1 ;; \ + esac \ + && SETUPTOOLS_SCM_PRETEND_VERSION_FOR_LMCACHE="${LMCACHE_VERSION}" \ + /source-bundles/uv run --no-project --python /opt/venv/bin/python \ + /opt/venv/bin/python -m pip wheel --no-build-isolation --no-deps --wheel-dir /artifacts . \ + && /opt/venv/bin/python /build-tools/pack_lmcache_native_reuse.py \ + --source /tmp/lmcache-source --reference-source /opt/lmcache/source \ + --reference-package /opt/venv/lib/python3.12/site-packages/lmcache \ + --wheel-dir /artifacts --mode "${LMCACHE_NATIVE_MODE}" \ + && install -Dm755 /opt/lmcache/lib/liblmcache_cumem_shareable.so /artifacts/liblmcache_cumem_shareable.so \ + && cd /artifacts && sha256sum *.whl *.so > SHA256SUMS + +FROM ${BASE_IMAGE} +ARG SOURCE_LOCK_SHA256 +ARG CACHE_FINGERPRINT +# One source installation layer above the immutable flattened runtime. +RUN --mount=type=bind,from=source_bundles,source=/,target=/source-bundles,ro \ + --mount=type=bind,from=native_artifact,source=/,target=/flashkda-artifacts,ro \ + --mount=type=bind,from=lmcache-build,source=/artifacts,target=/lmcache-artifacts,ro \ + --mount=type=bind,from=flashinfer-wheels,source=/opt/flashinfer-wheels,target=/flashinfer-wheels,ro \ + --mount=type=bind,source=.,target=/build-inputs,ro \ + bash /build-inputs/install_glm53_source_locked.sh + +ENV LMCACHE_EXPECTED_SOURCE_ROOT=/opt/venv/lib/python3.12/site-packages/lmcache \ + VLLM_DEFAULT_MOE_BACKEND=b12x \ + LOCAL_INFERENCE_CACHE_FINGERPRINT=${CACHE_FINGERPRINT} \ + XDG_CACHE_HOME=/cache/jit/${CACHE_FINGERPRINT} \ + VLLM_CACHE_ROOT=/cache/jit/${CACHE_FINGERPRINT}/vllm \ + VLLM_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/vllm \ + TRITON_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/triton \ + TORCHINDUCTOR_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/torchinductor \ + CUTE_DSL_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/cute-dsl \ + B12X_CUTE_COMPILE_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/b12x/cute \ + B12X_COMPILE_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/b12x/compile \ + SPARKINFER_COMPILE_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/b12x/compile \ + CUDA_CACHE_PATH=/cache/jit/${CACHE_FINGERPRINT}/cuda \ + VLLM_GLM53_L2_PREFETCH=1 VLLM_GLM53_L2_PREFETCH_PERSIST_MB=0 \ + VLLM_GLM53_DFLASH_ATTN=1 VLLM_CAUSAL_CONV1D_UPDATE_HOIST=1 \ + VLLM_GLM53_KDA_GATE_SIDE_STREAM=1 VLLM_GLM53_MTP_DRAFT_HEAD=nvfp4 \ + VLLM_MXFP8_LM_HEAD=0 VLLM_MTP_NVFP4_LM_HEAD=0 VLLM_LM_HEAD_A16=1 \ + VLLM_USE_FLASHINFER_SAMPLER=1 VLLM_PCIE_TWOSHOT_ALLREDUCE_MAX_SIZE=768KB \ + VLLM_GLM53_ONLINE_DENSE_MXFP8=0 VLLM_DISABLE_SHARED_EXPERTS_STREAM=0 \ + VLLM_DISABLED_KERNELS=MarlinFP8ScaledMMLinearKernel \ + B12X_DYNAMIC_SPLIT_ROUTE_COMPUTE=1 B12X_DYNAMIC_DIRECT_EXPERT_SCALES=1 \ + B12X_DYNAMIC_SPLIT_LOW_SMEM=1 B12X_DYNAMIC_SKIP_SPLIT_BARRIER_RESET=1 \ + B12X_DYNAMIC_SPLIT_FAST_PREPARE=1 B12X_DYNAMIC_WORK_SOURCE=persistent_grid \ + B12X_DYNAMIC_SPLIT_COMPUTE_MAC=224 B12X_PCIE_ONESHOT_THREADS=512 \ + B12X_PCIE_ONESHOT_BLOCK_LIMIT=4 B12X_PCIE_ONESHOT_PDL=1 B12X_MHC_PDL=1 \ + NCCL_MIN_NCHANNELS=16 NCCL_MAX_NCHANNELS=16 NCCL_BUFFSIZE=2097152 \ + OMP_NUM_THREADS=1 MAX_NUM_BATCHED_TOKENS=4096 MAX_CUDAGRAPH_CAPTURE_SIZE=256 \ + CUDAGRAPH_CAPTURE_SIZES="1 2 4 8 16 32 40 48 64 96 128 192 256" \ + PREFILL_SCHEDULE_INTERVAL=1 FAIRNESS_ENGINE=compute_share \ + PREFILL_COMPUTE_SHARE=0.4 GLM53_KDA_PREFILL_BACKEND=flashkda +LABEL local-inference.status="research-only" \ + local-inference.runtime.source-lock.path="/opt/glm53-flash/source.lock" \ + local-inference.runtime.source-lock.sha256="${SOURCE_LOCK_SHA256}" \ + local-inference.runtime.cache-fingerprint="${CACHE_FINGERPRINT}" \ + local-inference.release.overlay2-rootfs-layers="2" +WORKDIR / +ENTRYPOINT ["/usr/local/bin/serve-glm53-flash.sh"] +CMD [] diff --git a/recipes/glm53/Dockerfile.glm53-managed-checkpoint-native b/recipes/glm53/Dockerfile.glm53-managed-checkpoint-native new file mode 100644 index 00000000..f78158a0 --- /dev/null +++ b/recipes/glm53/Dockerfile.glm53-managed-checkpoint-native @@ -0,0 +1,33 @@ +# syntax=docker/dockerfile:1.7 + +ARG RUNTIME_IMAGE=voipmonitor/vllm@sha256:93ac5228f1cbde2182ca294d8b479259144742af2756a49ff207dd245429bf43 +FROM ${RUNTIME_IMAGE} AS native-build + +COPY --from=vllm_source / /tmp/vllm-source/ +COPY --from=cutlass_source / /tmp/cutlass-source/ +WORKDIR /tmp/vllm-source + +ENV VLLM_CUTLASS_SRC_DIR=/tmp/cutlass-source \ + TORCH_CUDA_ARCH_LIST=12.0a \ + CUDA_VISIBLE_DEVICES="" + +# FetchContent applies the source-locked FlashKDA patch. No external FlashKDA +# source override is supplied: this exercises the ordinary vLLM build contract. +RUN --mount=type=cache,id=glm53-managed-checkpoint-native,target=/tmp/flashkda-managed-build,sharing=locked \ + test -z "${FLASH_KDA_SRC_DIR:-}" \ + && cmake -S . -B /tmp/flashkda-managed-build -G Ninja \ + -DCMAKE_BUILD_TYPE=Release -DCMAKE_CUDA_ARCHITECTURES=120 \ + -DVLLM_TARGET_DEVICE=cuda \ + -DVLLM_PYTHON_EXECUTABLE=/opt/venv/bin/python -DNVCC_THREADS=2 \ + && cmake --build /tmp/flashkda-managed-build --target _flashkda_C --parallel 2 \ + && install -Dm755 /tmp/flashkda-managed-build/_flashkda_C.abi3.so \ + /artifacts/_flashkda_C.abi3.so \ + && git -C /tmp/flashkda-managed-build/_deps/flashkda-src rev-parse HEAD \ + > /artifacts/flashkda-base.commit \ + && sha256sum cmake/external_projects/patches/flashkda-packed-checkpoints.patch \ + > /artifacts/flashkda-patch.sha256 \ + && sha256sum /artifacts/_flashkda_C.abi3.so \ + > /artifacts/flashkda-extension.sha256 + +FROM scratch AS artifact +COPY --from=native-build /artifacts/ / diff --git a/recipes/glm53/Dockerfile.glm53-reasoning-default b/recipes/glm53/Dockerfile.glm53-reasoning-default new file mode 100644 index 00000000..1f98b8da --- /dev/null +++ b/recipes/glm53/Dockerfile.glm53-reasoning-default @@ -0,0 +1,19 @@ +# syntax=docker/dockerfile:1 +FROM voipmonitor/vllm@sha256:f5f121e37fd2afbb6f8f036e7eb627435cfb736de0a4420306dc2a25b6631669 + +ARG SOURCE_LOCK_SHA256 +# The reasoning default changes only request rendering. Target/draft weights, +# model kernels, cache implementation, and collective policies remain pinned. +RUN --mount=type=bind,source=serve-glm53-flash-nvfp4-dflash2.sh,target=/inputs/launcher,readonly \ + --mount=type=bind,source=reasoning-high.source.lock,target=/inputs/source.lock,readonly \ + test "$(sha256sum /inputs/source.lock | cut -d' ' -f1)" = "${SOURCE_LOCK_SHA256}" && \ + test "$(sha256sum /inputs/launcher | cut -d' ' -f1)" = \ + "$(awk -F= '$1 == "input.serve-glm53-flash-nvfp4-dflash2.sh.sha256" {print $2}' /inputs/source.lock)" && \ + install -m755 /inputs/launcher /usr/local/libexec/serve-glm53-flash-nvfp4-dflash2.sh && \ + install -m644 /inputs/source.lock /opt/glm53-flash/source.lock + +LABEL org.opencontainers.image.version="r28-high" \ + local-inference.release.name="jovian-judgement-community-20260908-r28-high" \ + local-inference.release.overlay2-rootfs-layers="3" \ + local-inference.runtime.default.reasoning-effort="high" \ + local-inference.runtime.source-lock.sha256="${SOURCE_LOCK_SHA256}" diff --git a/recipes/glm53/Dockerfile.glm53-scheduler-overlay b/recipes/glm53/Dockerfile.glm53-scheduler-overlay new file mode 100644 index 00000000..e78bc931 --- /dev/null +++ b/recipes/glm53/Dockerfile.glm53-scheduler-overlay @@ -0,0 +1,21 @@ +# syntax=docker/dockerfile:1 +FROM voipmonitor/vllm@sha256:f5f121e37fd2afbb6f8f036e7eb627435cfb736de0a4420306dc2a25b6631669 + +ARG SOURCE_LOCK_SHA256 +ARG CACHE_FINGERPRINT +RUN --mount=type=bind,source=.,target=/recipe,readonly \ + --mount=type=bind,from=source_bundles,target=/inputs,readonly \ + SOURCE_LOCK_SHA256=${SOURCE_LOCK_SHA256} \ + /opt/venv/bin/python /recipe/install_glm53_scheduler_overlay.py + +ENV LOCAL_INFERENCE_CACHE_FINGERPRINT=${CACHE_FINGERPRINT} \ + XDG_CACHE_HOME=/cache/jit/${CACHE_FINGERPRINT} \ + VLLM_CACHE_ROOT=/cache/jit/${CACHE_FINGERPRINT}/vllm \ + VLLM_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/vllm \ + TRITON_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/triton \ + TORCHINDUCTOR_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/torchinductor \ + CUTE_DSL_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/cute-dsl \ + B12X_CUTE_COMPILE_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/b12x/cute \ + B12X_COMPILE_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/b12x/compile \ + SPARKINFER_COMPILE_CACHE_DIR=/cache/jit/${CACHE_FINGERPRINT}/b12x/compile \ + CUDA_CACHE_PATH=/cache/jit/${CACHE_FINGERPRINT}/cuda diff --git a/recipes/glm53/Dockerfile.jovian-stable-native b/recipes/glm53/Dockerfile.jovian-stable-native new file mode 100644 index 00000000..07e6ffe9 --- /dev/null +++ b/recipes/glm53/Dockerfile.jovian-stable-native @@ -0,0 +1,25 @@ +# syntax=docker/dockerfile:1.7 +ARG RUNTIME_IMAGE=voipmonitor/vllm@sha256:93ac5228f1cbde2182ca294d8b479259144742af2756a49ff207dd245429bf43 +FROM ${RUNTIME_IMAGE} AS native-build +COPY --from=vllm_source / /tmp/vllm-source/ +COPY --from=cutlass_source / /tmp/cutlass-source/ +WORKDIR /tmp/vllm-source +ENV VLLM_CUTLASS_SRC_DIR=/tmp/cutlass-source \ + TRITON_KERNELS_SRC_DIR=/opt/glm53-flash/vllm/vllm/third_party/triton_kernels \ + TORCH_CUDA_ARCH_LIST=12.0a CUDA_VISIBLE_DEVICES="" +ARG BUILD_JOBS=8 +# The caller-owned DeepSeek query output changes the registered stable operator +# schema. Rebuild that extension; retain the separately qualified FlashKDA and +# MoE artifacts rather than substituting unrelated native implementations. +RUN --mount=type=cache,id=jovian-shared-stable-native-cu133,target=/tmp/stable-build,sharing=locked \ + cmake -S . -B /tmp/stable-build -G Ninja \ + -DCMAKE_BUILD_TYPE=Release -DCMAKE_CUDA_ARCHITECTURES=120 \ + -DVLLM_TARGET_DEVICE=cuda \ + -DVLLM_PYTHON_EXECUTABLE=/opt/venv/bin/python -DNVCC_THREADS=2 \ + && cmake --build /tmp/stable-build --target _C_stable_libtorch --parallel "${BUILD_JOBS}" \ + && install -Dm755 /tmp/stable-build/_C_stable_libtorch.abi3.so \ + /artifacts/_C_stable_libtorch.abi3.so \ + && cp native-source.identity /artifacts/native-source.identity \ + && cd /artifacts && sha256sum _C_stable_libtorch.abi3.so > SHA256SUMS +FROM scratch AS artifact +COPY --from=native-build /artifacts/ / diff --git a/recipes/glm53/README.md b/recipes/glm53/README.md new file mode 100644 index 00000000..ba106e66 --- /dev/null +++ b/recipes/glm53/README.md @@ -0,0 +1,401 @@ +# Source-locked Jovian Judgement serving image + +Status: **implemented; release qualification is tracked separately**. + +The source references below include the corrected NVFP4 SwiGLU clamp. The +published R34 image does not: do not use its B12X revision as a deployment +baseline. See the [R34 preservation and correctness audit](r34-preservation.md) +for immutable artifact identities, the GPU reproducer and qualification limits. + +The build installs complete committed vLLM, B12X and LMCache Git trees over an +immutable CUDA 13.3/PyTorch 2.13 runtime. It reuses source-compatible LMCache +native extensions and installs a source-authenticated FlashKDA operator with packed recurrent +checkpoint exports. The result has two filesystem layers; serving does not +require source-code bind mounts. + +The runtime foundation supplies compatible PyTorch, CUDA, NCCL, InstantTensor +and unchanged native binaries. The stable vLLM extension is rebuilt together +with its Python callers because the DeepSeek query-output operator has a +caller-owned output ABI. FlashInfer Python/JIT wheels come from the immutable +DS4 Vision artifact identified in the source lock. FlashKDA remains separately +authenticated. Replacing any native dependency requires serving qualification. + +## Build inputs + +For a pinned branch-plus-PR reconstruction, use the +[public review composition contract](review-composition.md). It verifies all +three component trees from public review heads without private resolution +patches. The source table below identifies those composed trees. Historical +R34 identities are recorded separately in the preservation audit; do not mix +components from the two compositions. + +Provide clean, complete Git checkouts as `vllm-source`, `b12x-source` and +`lmcache-source`. Use the release's published source refs and revisions, not +moving branch heads when reproducing a measured image. The image's +`/opt/glm53-flash/source.lock` records repository and package trees, build-input +hashes, Git-bundle hashes and native-artifact identity. + +Host requirements: Git, Docker Buildx and `uv`; no host PyTorch or active GPU +is needed for image construction. Compilation uses the runtime foundation. + +The GLM/Qwen/DS4 shared serving artifact uses the following published integration +refs. They preserve contributor history and integration resolutions; simply +merging arbitrary PR heads or changing the declared merge order does not +reproduce the source contract. The pinned, ordered manifests reproduce these +trees including their required integration resolutions. +These source refs are **implemented**, not by themselves release approval. + +| Checkout | Repository | Commit | +|---|---|---| +| `vllm-source` | [voipmonitor/vllm](https://github.com/voipmonitor/vllm/tree/integration/jovian-reviewed-sources-20260911) | `de982a50c6a3e4718e5cf9f00423a92192718da1` | +| `b12x-source` | [voipmonitor/b12x](https://github.com/voipmonitor/b12x/tree/integration/jovian-reviewed-sources-20260911) | `98086604c86ec1e78977e5023ce282ecb97ab8a7` | +| `lmcache-source` | [local-inference-lab/LMCache](https://github.com/local-inference-lab/LMCache/tree/release/jovian-fp4-fs-ledger-r33-20260910) | `29bc5a2efde737c436b04499eb62cd1776cebeec` | + +Clone each linked branch into its checkout directory, then verify `git rev-parse HEAD` +against the table. Complete checkouts are required; do not use shallow clones +for the source bundles. The package trees and compiled FlashKDA artifact are +reproducible inputs; OCI timestamps and archive metadata are not promised to +produce a bit-identical image digest on a different builder. + +From this recipe directory, build the FlashKDA operator: + +```bash +git clone --depth 1 --branch v4.4.2 https://github.com/NVIDIA/cutlass.git cutlass-source +docker buildx build \ + --build-context vllm_source=./vllm-source \ + --build-context cutlass_source=./cutlass-source \ + --output type=local,dest=./flashkda-artifact \ + -f Dockerfile.glm53-managed-checkpoint-native . +``` + +The native build resolves the FlashKDA revision and CUTLASS submodule declared +by vLLM, applies its recorded checkpoint patch, and exports the binary with +source and checksum records. Do not supply an unpatched `FLASH_KDA_SRC_DIR`. + +Build the stable extension from the same committed vLLM tree: + +```bash +uv run --no-project --python 3.12 build_jovian_stable_native.py \ + --vllm ./vllm-source --cutlass ./cutlass-source --output ./stable-artifact +cp ./stable-artifact/_C_stable_libtorch.abi3.so \ + ./stable-artifact/native-source.identity ./flashkda-artifact/ +``` + +The source identities must match the vLLM `csrc`, `cmake`, and `CMakeLists.txt` +objects in the final source bundle. The source table includes the DS4 +query-output ABI and serving contracts; a binary from a different caller ABI +is not a compatible substitute. + +Freeze source bundles and install them: + +```bash +uv run --no-project --python 3.12 prepare_glm53_source_bundles.py \ + --vllm ./vllm-source --b12x ./b12x-source --lmcache ./lmcache-source \ + --vllm-repository https://github.com/voipmonitor/vllm.git \ + --b12x-repository https://github.com/voipmonitor/b12x.git \ + --lmcache-repository https://github.com/local-inference-lab/LMCache.git \ + --native-artifact ./flashkda-artifact --uv "$(command -v uv)" \ + --lmcache-native-mode reuse-all \ + --output ./source-bundles \ + --release-name jovian-judgement-community-source-locked \ + --release-version source-locked +bash build_glm53_cache_contract_image.sh \ + ./source-bundles ./flashkda-artifact local/glm53:source-locked +``` + +The bundle directory must not already exist. Uncommitted serving changes, +bundle/tree mismatches and native-patch mismatches fail the build. Git bundles +preserve contributor history; no source squashing or filesystem overlays are +used to impersonate a committed tree. + +Repository labels come from the manifest. The optional `--*-repository` +arguments identify the published composition repository rather than inferring +its owner from the package name; without them, the preparer uses the checkout's +`origin`. URLs containing credentials are rejected. A recorded URL is not +proof of publication: verify that it serves the locked commit before release. +Native compilation archives the committed CUTLASS tree and records its actual +commit and tree instead of inferring a version from the build instructions. + +`source_locked_image_labels.py` clears inherited serving claims and replaces +them from the authenticated manifest. It preserves metadata for unchanged +runtime dependencies. An implementation label is not evidence of qualification. + +Run the CPU-only installation and metadata contracts with: + +```bash +uv run --no-project --with pytest python -m pytest -q tests/test_build_contract.py +``` + +## Serving contract + +### GLM cache HTTP binding + +Status: **implemented**. `LMCACHE_HTTP_HOST` selects the LMCache HTTP bind +address; the default remains `127.0.0.1`. Use `0.0.0.0` for pod-IP metrics +scraping only behind a trusted-network policy or authenticated proxy: this +listener also exposes administrative APIs. Readiness probes use a concrete +address compatible with the selected bind, including IPv6. + +Use this explicit setting for HTTP binding. The GLM wrappers do not interpret +extra arguments as shell commands or permit them to override checkpoint +transport, shared-memory ownership, or readiness addresses. + +### LMCache checkpoint publication, RAM retention and server arguments + +Status: **implemented**; GPU qualification must identify the tested image. + +Concurrent request-boundary generations may share immutable attention pages. +If another producer is writing a shared page, checkpoint admission returns a +metadata-only `busy` response. The worker-owned background transfer retries +with bounded backoff, without blocking the model or the LMCache RPC handlers. +A committed page needs no additional copy; an aborted owner's page can be +reserved by the waiting producer. Capacity and cancellation failures remain +safe misses, and publication still requires every rank's acknowledgement. +This behavior is implemented by Derek Yates's +[LMCache #65](https://github.com/local-inference-lab/LMCache/pull/65). +Run workers and the sidecar from the same image so both understand the response. + +`LMCACHE_L2_PREFETCH_POLICY` accepts `retain` (default) or `default`. +`retain` keeps objects loaded from the filesystem tier in the bounded L1 RAM +pool after readers finish. They remain eligible for ordinary eviction. +When filesystem storage is enabled, retained restores can synchronously evict +unlocked LRU objects to obtain staging space. This remains write-through: +restores do not switch the cache into synchronous writeback. Active read/write +owners remain protected, and insufficient reclaimable capacity remains a safe miss. +`default` releases temporary loaded objects when their final reader finishes. +This setting does not control the GPU hardware L2 prefetcher. + +`LMCACHE_SERVER_EXTRA_ARGS` appends whitespace-separated server arguments, +for example `--max-cpu-workers 4`. Arguments are literal: shell expansion, +quote interpretation, and multiline values are unsupported. Quoted array +expansion prevents wildcard characters from becoming filenames. Additional +scalar options take precedence over earlier defaults when the server accepts +repetition. Launcher-managed transport, identity, geometry, prefetch policy, +and listener options must use their dedicated settings instead. + +The interface derives from Tim Rice's +[launcher proposal](https://github.com/local-inference-lab/rtx6kpro/pull/100), +with literal argument handling and explicit protection for launcher-owned +configuration. + +LMCache packaging records `lmcache.native.mode` in the source lock. +`reuse-all` requires identical native sources, Rust sources, build policy and +requirements before copying the compiled payload. `cpu-rebuild-cuda-reuse` +builds all three common C++ extensions with `NO_GPU_EXT=1`, then copies only the +CUDA extension from the immutable `lmcache.native.artifact.image`. Every native +input outside the two CPU implementation directories must match that reference, +including shared headers and build policy. The platform wheel's tags and RECORD +describe the combined payload. The build-stage image does not contribute its +layers to the serving image, which retains two filesystem layers. + +The filesystem connector reconciles missing objects during idempotent deletion +and protects keys with pending native stores, following Derek Yates's +[LMCache #67](https://github.com/local-inference-lab/LMCache/pull/67). Bounded +object paths, readable legacy paths and real filesystem errors remain supported. +The native donor identified in `lmcache.native.artifact.image` contains this +filesystem correction. `reuse-all` retains that compiled implementation after +the source comparison. A donor without the correction requires +`cpu-rebuild-cuda-reuse`; copying only Python is insufficient. + +The image sets `VLLM_DEFAULT_MOE_BACKEND=b12x`. Native `vllm serve` and Python +`KernelConfig()` therefore select B12X when no MoE backend is supplied. An +explicit `--moe-backend auto`, another backend, or a nested kernel configuration +overrides that default. The GLM wrapper also supports `MOE_BACKEND`; its default +is B12X. This policy does not replace attention or sampler backends, nor explicit +speculative-model backend choices. Outside the image, vLLM retains `auto` when +`VLLM_DEFAULT_MOE_BACKEND` is unset. Unsupported explicit B12X configurations +fail normally rather than silently selecting another implementation. + +Persistent LMCache namespaces resolve the effective positional or `MODEL` +checkpoint and its immutable model/draft identity in both request-boundary and +aligned modes. Hub revisions are pinned; local weights and configuration are +content-hashed before serving. Aligned transfers use that identity only for +the filesystem namespace, without switching to the semantic connector. +Dry-run performs no model I/O and labels unavailable identities as unresolved. + +### Publisher sampling defaults + +Status: **implemented**, with CPU launcher-precedence tests. GPU quality and +performance measurements identify the sampling parameters actually supplied; +changing a default is not evidence of a model-quality repair. + +The GLM launcher supplies `--override-generation-config +'{"temperature":1.0,"top_p":0.95}'` because a quantized checkpoint can omit +the [Z.ai generation defaults](https://huggingface.co/zai-org/GLM-5.3-Flash/blob/main/generation_config.json). +The DS4 text and Vision launcher uses the same values for its agentic serving +profile, following the [DeepSeek text deployment recommendation](https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731#how-to-run-locally) +and [Vision agent evaluation settings](https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp). +For non-agentic DS4 use, DeepSeek recommends top-p 1.0 instead. + +Explicit API request parameters override these server defaults. Supplying +`--generation-config`, `--override-generation-config` (including dotted fields) +or a native `--config` file selects operator policy instead of the launcher +preset. It is forwarded unchanged; the launcher does not add a competing JSON +object. DS4 also recognizes these options in `EXTRA_VLLM_ARGS`. + +The separately documented Qwen native launch uses its checkpoint's generation +configuration: temperature 1.0, top-p 0.95 and top-k 20. These match the +[Qwen thinking-mode defaults](https://huggingface.co/Qwen/Qwen3.8-Flash-Next/blob/main/generation_config.json) +and are present in the qualified NVFP4 revision +`b797d2e1160b9596b2570e56c1d3590faa09d4ed`. Qwen's non-thinking profile has +different recommendations; disabling thinking alone does not automatically +change request sampling. A command that bypasses the model launcher must set +its own generation policy or retain the publisher's checkpoint metadata. + +GLM's community chat defaults remain `reasoning_effort=high` and +`clear_thinking=false`. Sampling defaults do not clear historical reasoning, +truncate prompts, change the target KV dtype or quantize model weights. + +### DeepSeek V4 model profiles + +Status: **implemented; combined-image serving qualification is tracked separately**. +Use `--entrypoint /usr/local/bin/serve-ds4-jovian.sh` or +`ds4-jovian.compose.yml`. This selects the same installed packages as GLM, +with B12X attention/W4A8 MoE and DeepGEMM dense projections. It does not change +GLM or Qwen backend selection. + +`DS4_MODEL_VARIANT=text` selects `deepseek-ai/DeepSeek-V4-Flash-0731`, fixed +probabilistic DSpark K5 and eight request slots. `DS4_MODEL_VARIANT=vision` +selects `deepseek-ai/DeepSeek-V4-Flash-Vision-Exp`, K3 and four request slots. +Both default to TP2/DCP1, FP8 compressed KV, buffered InstantTensor, +`FULL_AND_PIECEWISE`, and a 4096-token scheduler budget. `MODEL` can name a +local checkpoint or a Hugging Face repository. The native launcher reports +the resolved model revision and serving command. + +The image has GLM-specific graph and CPU environment defaults. The DS4 +entrypoint deliberately replaces those with `DS4_OMP_NUM_THREADS=2`, +`DS4_MAX_CUDAGRAPH_CAPTURE_SIZE=auto`, and +`DS4_CUDAGRAPH_CAPTURE_SIZES=default`. Automatic caps are 48 rows for text K5 +and 16 for Vision K3. Use these DS4-prefixed settings to override that profile; +changing the GLM environment variables does not change the DS4 graph contract. + +GPU-local caching is the default. `LMCACHE_MODE=ram|disk` enables worker-owned +asynchronous SHM transfer with a CPU-only sidecar, a 24 GiB L1 pool and +4096-token objects. `ram` does not enable filesystem storage. `disk` uses +`LMCACHE_L2_PATH` and `LMCACHE_L2_GB`. Select independent cache ports/namespaces +when serving multiple models. DS4 engine-driven TP2 with a model limit of at +least one million tokens defaults to utilization 0.970; GPU-only uses 0.975. +These are reference deployment profiles for 96 GiB GPUs, not universal memory +capacity or topology guarantees. Validate admission on the deployment hardware. + +Use an empty external-cache namespace when adopting the paged-gather metadata +correction in LMCache #50. The source revision in the table keeps every batch's +pinned block-ID snapshot immutable until its asynchronous copy completes. +Payloads exported by a build without that correction can contain bytes from +different GPU pages while retaining valid storage keys. DS4 filesystem keys +do not automatically invalidate those payloads on a source update. Use a +separate Docker cache volume or L2 directory; do not delete another service's +cache or treat a successful read as proof that an existing payload is valid. + +### Python scheduler overlay + +Status: implemented; GPU qualification is recorded per image in the linked +model runbook. `Dockerfile.glm53-scheduler-overlay` applies scheduler Python +sources and launcher controls to the immutable R28 FP8 image without replacing +CUDA, PyTorch, B12X, FlashKDA, NCCL, or LMCache. It produces three filesystem +layers. The installed Git history remains complete; the incremental bundle +requires the parent image's authenticated source objects. + +Extract the parent manifest and prepare the overlay from a clean vLLM checkout: + +```bash +git -C vllm-source fetch https://github.com/voipmonitor/vllm.git \ + integration/glm-scheduler-hardening-20260908 +git -C vllm-source checkout --detach 9ff42d83938e74018f9c255e8cfa7ca6df6921b0 +docker run --rm --entrypoint /bin/cat \ + voipmonitor/vllm@sha256:f5f121e37fd2afbb6f8f036e7eb627435cfb736de0a4420306dc2a25b6631669 \ + /opt/glm53-flash/source.lock > r28.source.lock +uv run --no-project prepare_glm53_scheduler_overlay.py \ + --vllm ./vllm-source --parent-lock ./r28.source.lock \ + --uv "$(command -v uv)" --output ./scheduler-bundles \ + --release-name jovian-judgement-community-20260908-r28.1 \ + --release-version r28.1 +bash build_glm53_scheduler_overlay.sh ./scheduler-bundles local/glm53:scheduler +``` + +The preparer rejects changes outside the scheduler and its tests. The installer +verifies the parent manifest, source trees, input hashes, and FlashKDA binary. +Generated vLLM version metadata identifies the installed Python source; native +component identities remain those in the manifest. A changed source lock +produces safe external-checkpoint misses across image versions. It does not +reinterpret persistent checkpoints written by another runtime identity. + +### Scheduler controls + +The image retains fixed prefill compute share `0.4`, schedule interval `1`, and +one prefill lane. Interleaving is opt-in. Explicit CLI values override the +corresponding environment variable and are emitted only once. + +| Environment | vLLM option | Values | +|---|---|---| +| `PREFILL_COMPUTE_SHARE` | `--prefill-compute-share` | `auto` or finite float strictly between 0 and 1 | +| `PREFILL_COMPUTE_HALF_LIFE` | `--prefill-compute-half-life` | `smooth`, `responsive`, or positive finite seconds; requires auto share | +| `MAX_PARALLEL_PREFILLS` | `--max-parallel-prefills` | `auto` or positive integer | +| `PREFILL_POLICY` | `--prefill-policy` | `round-robin` or `decode-aware` | +| `DECODE_REFILL_TARGET` | `--decode-refill-target` | `auto` or positive integer | + +Automatic lane count is `min(4, max_num_seqs)`, independently of attention, +recurrent, or LMCache object geometry. The scheduler token budget still limits +each forward pass. With automatic refill, the refill target follows the +effective lane count. A numeric fixed share rejects any half-life setting. +Compute-share scheduling requires interval `1`. + +`FAIRNESS_ENGINE=none` suppresses the inherited environment share; an explicit +CLI share remains authoritative. `FAIRNESS_ENGINE=compute_share` is supported +for compatibility, while `micro_slicing` is rejected. `--help` lists the +controls without loading models. `CACHE_MODE=vram DRY_RUN=1` prints the full +command. Unit tests run with `uv run --no-project --with pytest pytest -q tests`. + +### Models and cache + +The installed launcher is `/usr/local/bin/serve-glm53-flash.sh`. Model defaults +are `local-inference-lab/GLM-5.3-Flash-NVFP4` and +`local-inference-lab/GLM-5.3-Flash-DFlash2`; revisions are resolved at startup. +The latter checkpoint contains offline MXFP8 draft weights. A changed model or +runtime identity cannot reuse an incompatible external recurrent checkpoint. + +The GLM launcher supplies `--default-chat-template-kwargs +'{"reasoning_effort":"high","clear_thinking":false}'`. +A chat request can override the reasoning default with +`"reasoning_effort":"max"`, `"high"`, or `"low"`; this does not change +`max_tokens` or the parser. Without a server/request override, the checkpoint's +chat template selects `max`. Direct `vllm serve` commands bypass this launcher's +defaults and must provide the option themselves. + +The `clear_thinking=false` agent profile follows the +[model author's preserved-thinking guidance](https://docs.z.ai/guides/capabilities/thinking-mode#preserved-thinking). +Clients must return the complete, ordered assistant reasoning with the message +history. The template retains that reasoning at its original assistant turn; +it does not insert historical reasoning into the generated turn. The setting +cannot recover reasoning omitted by the client. + +Removing completed-turn reasoning changes the rendered token prefix. A cached +response containing that reasoning cannot be reused past the first changed +token; instruction and matching input-prefix checkpoints remain usable. +Chat applications intentionally omitting completed-turn reasoning can send +`"chat_template_kwargs":{"clear_thinking":true}`. An explicit server CLI +`--default-chat-template-kwargs` object replaces the launcher's object instead +of being duplicated. These defaults apply to the GLM entrypoint, not DS4 or +Qwen native serving commands. + +The production scheduler budget is 4096 tokens, OMP threads1, NCCL channels16 +with 2 MiB buffers, and `FULL_AND_PIECEWISE` CUDA graphs. GPU-local attention +pages default to2048 tokens. Target operations use B12X; FlashKDA supplies +checkpoint-producing recurrent prefill. The private MTP vocabulary projection +is NVFP4, while the target vocabulary projection remains BF16. + +Modes use `SPECULATOR=mtp MTP_DEPTH=0`, `SPECULATOR=mtp MTP_DEPTH=3`, or +`SPECULATOR=dflash2 DFLASH_DEPTH=7`. External cache is opt-in with +`CACHE_MODE=lmcache`; its default transfer is `engine_driven`. Semantic +checkpoints use immutable all-rank storage and asynchronous shared-memory +copies performed by vLLM workers, without a sidecar CUDA context. +Checkpoint payload namespaces are scoped to rank and storage group so that +unlocked pages remain evictable under RAM pressure. Complete manifests and +transfer leases enforce atomic retrieval. Version-1 semantic payloads are +incompatible with these version-2 keys and produce safe cache misses. + +Mode-specific deployment instructions and measured conditions belong in the +[GLM-5.3-Flash wiki](https://github.com/local-inference-lab/rtx6kpro/blob/master/models/glm-5.3-flash.md). +Qwen source may be included in the selected Git trees, but this launcher serves +GLM. A source build alone establishes neither Qwen runtime qualification nor +NVFP4 target-KV qualification. diff --git a/recipes/glm53/build_glm53_cache_contract_image.sh b/recipes/glm53/build_glm53_cache_contract_image.sh new file mode 100644 index 00000000..4b353e2a --- /dev/null +++ b/recipes/glm53/build_glm53_cache_contract_image.sh @@ -0,0 +1,36 @@ +#!/usr/bin/env bash +set -euo pipefail + +if (($# != 3)); then + echo "usage: $0 " >&2 + exit 2 +fi +readonly bundles=$(realpath -e "$1") native=$(realpath -e "$2") tag=$3 +readonly directory=$(cd -- "$(dirname -- "$0")" && pwd) +readonly lock="$bundles/source.lock" +value() { awk -F= -v key="$1" '$1 == key {sub(/^[^=]*=/, ""); print; found=1} END {if (!found) exit 1}' "$lock"; } +readonly fingerprint=$(value runtime.cache.fingerprint) +readonly lock_sha=$(sha256sum "$lock" | cut -d' ' -f1) +labels=() +mapfile -d '' -t source_labels < <( + docker image inspect "$(value runtime.base.image)" | \ + "$bundles/uv" run --no-project --python 3.12 "$directory/source_locked_image_labels.py" \ + --source-lock "$lock" +) +(( ${#source_labels[@]} > 0 )) || { echo 'Image metadata generation failed.' >&2; exit 1; } +for label in "${source_labels[@]}"; do labels+=(--label "$label"); done +docker buildx build --load --progress plain --tag "$tag" \ + --build-arg "SOURCE_LOCK_SHA256=$lock_sha" \ + --build-arg "LMCACHE_NATIVE_IMAGE=$(value lmcache.native.artifact.image)" \ + --build-arg "LMCACHE_COMMIT=$(value lmcache.commit)" \ + --build-arg "LMCACHE_TREE=$(value lmcache.tree)" \ + --build-arg "LMCACHE_BUNDLE_SHA256=$(value lmcache.bundle.sha256)" \ + --build-arg "LMCACHE_VERSION=$(value lmcache.version)" \ + --build-arg "LMCACHE_NATIVE_MODE=$(value lmcache.native.mode)" \ + --build-arg "UV_SHA256=$(value build.uv.sha256)" \ + --build-arg "CACHE_FINGERPRINT=$fingerprint" "${labels[@]}" \ + --build-context "source_bundles=$bundles" --build-context "native_artifact=$native" \ + -f "$directory/Dockerfile.glm53-cache-contracts" "$directory" +test "$(docker image inspect --format '{{len .RootFS.Layers}}' "$tag")" = 2 +test "$(docker image inspect --format '{{index .Config.Labels "local-inference.runtime.source-lock.sha256"}}' "$tag")" = "$lock_sha" +docker image inspect --format '{{.Id}}' "$tag" diff --git a/recipes/glm53/build_glm53_scheduler_overlay.sh b/recipes/glm53/build_glm53_scheduler_overlay.sh new file mode 100644 index 00000000..057219eb --- /dev/null +++ b/recipes/glm53/build_glm53_scheduler_overlay.sh @@ -0,0 +1,33 @@ +#!/usr/bin/env bash +set -euo pipefail + +if (($# != 2)); then + echo "usage: $0 " >&2 + exit 2 +fi +readonly bundles=$(realpath -e "$1") tag=$2 +readonly recipe=$(cd -- "$(dirname -- "$0")" && pwd) +readonly lock="$bundles/source.lock" +value() { awk -F= -v key="$1" '$1 == key {sub(/^[^=]*=/, ""); print; found=1} END {if (!found) exit 1}' "$lock"; } +readonly parent=$(value runtime.parent.release-image) +readonly lock_sha=$(sha256sum "$lock" | cut -d' ' -f1) +labels=() +mapfile -d '' -t source_labels < <( + docker image inspect "$parent" | "$bundles/uv" run --offline --no-project \ + --python 3.12 "$recipe/source_locked_image_labels.py" --source-lock "$lock" +) +((${#source_labels[@]} > 0)) || { echo 'Image label generation failed.' >&2; exit 1; } +for label in "${source_labels[@]}"; do labels+=(--label "$label"); done +labels+=(--label "local-inference.runtime.parent-image=$parent" + --label "local-inference.runtime.default.reasoning-effort=high" + --label "local-inference.vllm.version=$(value vllm.version)" + --label 'local-inference.scheduler.max-parallel-prefills=1' + --label 'local-inference.release.rootfs-format=two-layer FP8 runtime plus Python scheduler and launcher overlay') +docker buildx build --load --progress plain --tag "$tag" \ + --build-arg "SOURCE_LOCK_SHA256=$lock_sha" \ + --build-arg "CACHE_FINGERPRINT=$(value runtime.cache.fingerprint)" \ + "${labels[@]}" --build-context "source_bundles=$bundles" \ + -f "$recipe/Dockerfile.glm53-scheduler-overlay" "$recipe" +test "$(docker image inspect --format '{{len .RootFS.Layers}}' "$tag")" = 3 +test "$(docker image inspect --format '{{index .Config.Labels "local-inference.runtime.source-lock.sha256"}}' "$tag")" = "$lock_sha" +docker image inspect --format '{{.Id}}' "$tag" diff --git a/recipes/glm53/build_jovian_stable_native.py b/recipes/glm53/build_jovian_stable_native.py new file mode 100644 index 00000000..fee08a88 --- /dev/null +++ b/recipes/glm53/build_jovian_stable_native.py @@ -0,0 +1,105 @@ +"""Build the stable vLLM extension from a committed source snapshot.""" + +from __future__ import annotations + +import argparse +import subprocess +import tempfile +from pathlib import Path + + +def git(root: Path, *args: str) -> str: + return subprocess.check_output(["git", "-C", str(root), *args], text=True).strip() + + +def cutlass_identity(root: Path) -> dict[str, str]: + if git(root, "status", "--porcelain"): + raise ValueError("Native compilation requires committed CUTLASS sources") + return { + "cutlass.commit": git(root, "rev-parse", "HEAD"), + "cutlass.tree": git(root, "rev-parse", "HEAD^{tree}"), + } + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + for name in ("vllm", "cutlass", "output"): + parser.add_argument(f"--{name}", type=Path, required=True) + parser.add_argument("--jobs", type=int, default=8) + args = parser.parse_args() + if git(args.vllm, "status", "--porcelain"): + raise ValueError("Native compilation requires committed vLLM sources") + if args.jobs < 1 or args.output.exists(): + raise ValueError("Use positive build jobs and an absent output directory") + cutlass = cutlass_identity(args.cutlass) + recipe = Path(__file__).resolve().parent + with tempfile.TemporaryDirectory(prefix="jovian-native-") as temporary: + root = Path(temporary) + archive = root / "source.tar" + subprocess.run( + ["git", "-C", str(args.vllm), "archive", "HEAD", f"--output={archive}"], + check=True, + ) + source = root / "source" + source.mkdir() + subprocess.run(["tar", "-xf", str(archive), "-C", str(source)], check=True) + cutlass_archive = root / "cutlass.tar" + subprocess.run( + [ + "git", + "-C", + str(args.cutlass), + "archive", + cutlass["cutlass.commit"], + f"--output={cutlass_archive}", + ], + check=True, + ) + cutlass_source = root / "cutlass" + cutlass_source.mkdir() + subprocess.run( + ["tar", "-xf", str(cutlass_archive), "-C", str(cutlass_source)], + check=True, + ) + identity = {"vllm.commit": git(args.vllm, "rev-parse", "HEAD")} + for key, path in ( + ("csrc.tree", "csrc"), + ("cmake.tree", "cmake"), + ("cmakelists.blob", "CMakeLists.txt"), + ): + identity[f"vllm.{key}"] = git(args.vllm, "rev-parse", f"HEAD:{path}") + identity.update( + { + **cutlass, + "cuda.architecture": "120", + "native.target": "_C_stable_libtorch", + } + ) + (source / "native-source.identity").write_text( + "".join(f"{key}={value}\n" for key, value in identity.items()) + ) + subprocess.run( + [ + "docker", + "buildx", + "build", + "--progress", + "plain", + "--build-context", + f"vllm_source={source}", + "--build-context", + f"cutlass_source={cutlass_source}", + "--build-arg", + f"BUILD_JOBS={args.jobs}", + "--output", + f"type=local,dest={args.output.resolve()}", + "-f", + str(recipe / "Dockerfile.jovian-stable-native"), + str(recipe), + ], + check=True, + ) + + +if __name__ == "__main__": + main() diff --git a/recipes/glm53/ds4-jovian.compose.yml b/recipes/glm53/ds4-jovian.compose.yml new file mode 100644 index 00000000..89da4e92 --- /dev/null +++ b/recipes/glm53/ds4-jovian.compose.yml @@ -0,0 +1,55 @@ +# DeepSeek V4 text/Vision over the shared Jovian Judgement serving stack. +# MODEL_DIR is a locally staged checkpoint directory, mounted read-only. +# DS4_MODEL_VARIANT=vision selects Vision K3; text selects DSpark K5. +services: + ds4: + image: ${IMAGE:?Set IMAGE to the source-locked shared serving image} + container_name: ${NAME:-ds4-jovian} + entrypoint: [/usr/local/bin/serve-ds4-jovian.sh] + network_mode: host + ipc: host + init: true + restart: "no" + ulimits: + memlock: {soft: -1, hard: -1} + stack: 67108864 + environment: + CUDA_VISIBLE_DEVICES: "0,1" + MODEL: /model + DS4_MODEL_VARIANT: ${DS4_MODEL_VARIANT:-text} + PORT: ${PORT:-8000} + TP_SIZE: "2" + DCP_SIZE: "1" + BACKEND: ${BACKEND:-b12x-a8-dglin} + MAX_MODEL_LEN: ${MAX_MODEL_LEN:--1} + MAX_NUM_BATCHED_TOKENS: ${MAX_NUM_BATCHED_TOKENS:-4096} + GPU_MEMORY_UTILIZATION: ${GPU_MEMORY_UTILIZATION:-} + DS4_OMP_NUM_THREADS: ${DS4_OMP_NUM_THREADS:-2} + DS4_MAX_CUDAGRAPH_CAPTURE_SIZE: ${DS4_MAX_CUDAGRAPH_CAPTURE_SIZE:-auto} + DS4_CUDAGRAPH_CAPTURE_SIZES: ${DS4_CUDAGRAPH_CAPTURE_SIZES:-default} + GLOO_SOCKET_IFNAME: ${GLOO_SOCKET_IFNAME:-lo} + NCCL_SOCKET_IFNAME: ${NCCL_SOCKET_IFNAME:-lo} + NCCL_MIN_NCHANNELS: "16" + NCCL_MAX_NCHANNELS: "16" + NCCL_BUFFSIZE: "2097152" + LMCACHE_MODE: ${LMCACHE_MODE:-off} + LMCACHE_TRANSFER_MODE: ${LMCACHE_TRANSFER_MODE:-engine_driven} + LMCACHE_L1_GB: ${LMCACHE_L1_GB:-24} + LMCACHE_CHUNK_SIZE: "4096" + LMCACHE_SEPARATE_OBJECT_GROUPS: "1" + LMCACHE_PORT: ${LMCACHE_PORT:-5555} + LMCACHE_HTTP_PORT: ${LMCACHE_HTTP_PORT:-8099} + LMCACHE_L2_PATH: /cache/lmcache/ds4 + LMCACHE_L2_GB: ${LMCACHE_L2_GB:-256} + volumes: + - ${MODEL_DIR:?Set MODEL_DIR to a DeepSeek V4 checkpoint}:/model:ro + - ds4-cache:/cache + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: ["${GPU_0:-0}", "${GPU_1:-1}"] + capabilities: [gpu] +volumes: + ds4-cache: diff --git a/recipes/glm53/evidence/r34-preservation/artifact-summary.json b/recipes/glm53/evidence/r34-preservation/artifact-summary.json new file mode 100644 index 00000000..6c6da672 --- /dev/null +++ b/recipes/glm53/evidence/r34-preservation/artifact-summary.json @@ -0,0 +1,112 @@ +{ + "configuration_differences": {}, + "difference_counts": { + "b12x": 275, + "build_cache": 18807, + "git_provenance": 31, + "python_bytecode": 826, + "python_environment": 2, + "runtime_packaging": 2, + "vllm": 114 + }, + "environment_difference_hashes": { + "B12X_COMPILE_CACHE_DIR": { + "composition": "27f46020fd34240a1cb0ebd30abad4d00d775cd037fbf7880ce073a62330226c", + "r34": "08aeb13a1a72b27b8a7a40fc7867129972bbc0773e3f6a87851163b30a20f86f" + }, + "B12X_CUTE_COMPILE_CACHE_DIR": { + "composition": "d8f02bc216f2a3f92a44f4e0ad17a9a6924c73ad9af8a171f7f88c8d39c40bd9", + "r34": "3b1281a4a1f968ce467bec6f42dbe153940c049f26d3bb1c872a15d65386eaff" + }, + "CUDA_CACHE_PATH": { + "composition": "31ee71ed3cf5fd742b650e9a157a3b02745ad0324d16ac5a02b333303aaf1abd", + "r34": "c039afa0f9556ea3406191b2414996b93727aada193ac545b889e5e3c6484135" + }, + "CUTE_DSL_CACHE_DIR": { + "composition": "6d59cac25ec45b9349bd6e09cb1b500434f61be0dda0a5e434ad5791431368ee", + "r34": "74d43a397d0979ad20f66bf01b43699f71a1cdb3e79d202afeb6f19c04815ee0" + }, + "LOCAL_INFERENCE_CACHE_FINGERPRINT": { + "composition": "7bd4ce4bf06b3a9c0105c9439e3592905edbbc5a23652f06a2b78517525f1615", + "r34": "de4eb55836928881f9d04267fd56ef4c400df4bcee6f1341cdf0c684f1d7f366" + }, + "SPARKINFER_COMPILE_CACHE_DIR": { + "composition": "27f46020fd34240a1cb0ebd30abad4d00d775cd037fbf7880ce073a62330226c", + "r34": "08aeb13a1a72b27b8a7a40fc7867129972bbc0773e3f6a87851163b30a20f86f" + }, + "TORCHINDUCTOR_CACHE_DIR": { + "composition": "c2344d4d5d9985bc4c0f9e0eefecdce953825a7feed14a2473bdb8aa4668f1c1", + "r34": "77be359c6339f7932c2327e035abd6be340d552e53c37716693057c709b2358b" + }, + "TRITON_CACHE_DIR": { + "composition": "8c479d24391187ed97cd416b2ef19475e5d94f7ff43d47b8f495e92eb8520b15", + "r34": "192d18a5cf61cf6e004f6297280279a69cf667f2eb984e4ea38761db1d59ed97" + }, + "VLLM_CACHE_DIR": { + "composition": "30af5b23112c380ce2351130122b3ac8e558e7d28bf25494a77dbb2796280b60", + "r34": "c76ecfb2819d4c53e431d3854dafe4b34b603936681a1c71bb874d985c85895c" + }, + "VLLM_CACHE_ROOT": { + "composition": "30af5b23112c380ce2351130122b3ac8e558e7d28bf25494a77dbb2796280b60", + "r34": "c76ecfb2819d4c53e431d3854dafe4b34b603936681a1c71bb874d985c85895c" + }, + "XDG_CACHE_HOME": { + "composition": "40e81ba07b05f84f3eaff635e840a36066fabe50ae05a3cc5b68133dc96e74e0", + "r34": "8631dc8a1b9389303827ec1efbb9202c072f42b11992482ff1bc1cb278d0a252" + } + }, + "images": { + "composition": "sha256:d3508d5afd3c616db55efb5d18490e12370fdd0f9c6821f434136e4f4fc21c6a", + "r34": "sha256:eca2707b9ba1748196a0db72f659ebd1d2f4863cfde33d8da59924ea5983a19e" + }, + "scope": "Upper-layer content, modes, ownership and xattrs; timestamps excluded.", + "shared_runtime_base_layer": "sha256:cb220f94ebc323749ae8e960eecd0076d01b4a23ee7d2dabdb463b44fdc37d68", + "source_checks": { + "composition": { + "b12x": { + "commit": "98086604c86ec1e78977e5023ce282ecb97ab8a7", + "differences": [], + "submodules_not_recursed": [], + "tracked_files": 1015, + "tree": "d963f030c72a426753d2afa412b35d29eeb045f7" + }, + "lmcache": { + "commit": "29bc5a2efde737c436b04499eb62cd1776cebeec", + "differences": [], + "submodules_not_recursed": [], + "tracked_files": 2423, + "tree": "5a88a1ea9d2627c76288d056e7193f7454669b64" + }, + "vllm": { + "commit": "de982a50c6a3e4718e5cf9f00423a92192718da1", + "differences": [], + "submodules_not_recursed": [], + "tracked_files": 6870, + "tree": "6d5cc3490735a53e849a4f0909183cb5e3a22e8b" + } + }, + "r34": { + "b12x": { + "commit": "59d51a36a942d56a9c36265855cdc7856fa7712e", + "differences": [], + "submodules_not_recursed": [], + "tracked_files": 881, + "tree": "474364a1f7ae3e6d379be88a57ebbe02bb5f9dba" + }, + "lmcache": { + "commit": "29bc5a2efde737c436b04499eb62cd1776cebeec", + "differences": [], + "submodules_not_recursed": [], + "tracked_files": 2423, + "tree": "5a88a1ea9d2627c76288d056e7193f7454669b64" + }, + "vllm": { + "commit": "c496604123b1f4441007b952a7ee37ab12c8f6ad", + "differences": [], + "submodules_not_recursed": [], + "tracked_files": 6835, + "tree": "e5fbfbaf6815210b68daea1b0e09f373a2f6df1f" + } + } + } +} diff --git a/recipes/glm53/evidence/r34-preservation/composition-contracts-cpu.log b/recipes/glm53/evidence/r34-preservation/composition-contracts-cpu.log new file mode 100644 index 00000000..09d6fc3a --- /dev/null +++ b/recipes/glm53/evidence/r34-preservation/composition-contracts-cpu.log @@ -0,0 +1,2 @@ +............... [100%] +15 passed in 3.43s diff --git a/recipes/glm53/evidence/r34-preservation/composition-dispatch-gpu.log b/recipes/glm53/evidence/r34-preservation/composition-dispatch-gpu.log new file mode 100644 index 00000000..fd85f824 --- /dev/null +++ b/recipes/glm53/evidence/r34-preservation/composition-dispatch-gpu.log @@ -0,0 +1,2 @@ +.... [100%] +4 passed in 22.35s diff --git a/recipes/glm53/evidence/r34-preservation/composition-swiglu-gpu.log b/recipes/glm53/evidence/r34-preservation/composition-swiglu-gpu.log new file mode 100644 index 00000000..82ccb168 --- /dev/null +++ b/recipes/glm53/evidence/r34-preservation/composition-swiglu-gpu.log @@ -0,0 +1,2 @@ +. +1 passed in 13.80s diff --git a/recipes/glm53/evidence/r34-preservation/profile-preservation.json b/recipes/glm53/evidence/r34-preservation/profile-preservation.json new file mode 100644 index 00000000..9e07f5be --- /dev/null +++ b/recipes/glm53/evidence/r34-preservation/profile-preservation.json @@ -0,0 +1,23 @@ +{ + "b12x/policy/_profiles/data/nvidia.gb10.48sm.json.gz": { + "exceptions": [], + "field_counts": { + "release_specific_retained": 1, + "upstream_or_unchanged": 24 + } + }, + "b12x/policy/_profiles/data/nvidia.rtx.pro.6000.blackwell.json.gz": { + "exceptions": [], + "field_counts": { + "release_specific_retained": 1, + "upstream_or_unchanged": 24 + } + }, + "b12x/policy/_profiles/data/nvidia.rtx.pro.6000.blackwell.max-q.json.gz": { + "exceptions": [], + "field_counts": { + "release_specific_retained": 1, + "upstream_or_unchanged": 24 + } + } +} diff --git a/recipes/glm53/evidence/r34-preservation/r34-minimal-swiglu-gpu.log b/recipes/glm53/evidence/r34-preservation/r34-minimal-swiglu-gpu.log new file mode 100644 index 00000000..16845d9e --- /dev/null +++ b/recipes/glm53/evidence/r34-preservation/r34-minimal-swiglu-gpu.log @@ -0,0 +1,2 @@ +. +1 passed in 13.25s diff --git a/recipes/glm53/evidence/r34-preservation/r34-swiglu-gpu.log b/recipes/glm53/evidence/r34-preservation/r34-swiglu-gpu.log new file mode 100644 index 00000000..9deeddb4 --- /dev/null +++ b/recipes/glm53/evidence/r34-preservation/r34-swiglu-gpu.log @@ -0,0 +1,56 @@ +F +=================================== FAILURES =================================== +________ TestNvfp4SplitDispatch.test_swiglu_limit_changes_split_output _________ + +self = +domain = {'E': 8, 'K': 256, 'expert_tile_base': tensor([ 0, 1, 2, 3, 5, 6, 7, 9, 11], device='cuda:0', + dtype=torch.int32), 'intermediate_tiles': 1, ...} + + def test_swiglu_limit_changes_split_output(self, domain): + experts = _prepare_dispatch_experts(domain) + clamped_oracle = _dispatch_oracle(domain, swiglu_limit=2.0) + outputs = [] + for limit in (None, 2.0): + binding = make_tp_moe_fp4_binding( + a=domain["x"], + experts=experts, + topk_ids=domain["topk_ids"], + topk_weights=domain["topk_weights"], + input_scales_static=True, + fast_math=False, + swiglu_limit=limit, + ) + outputs.append(fused_moe.run(binding=binding).clone()) + torch.cuda.synchronize() + unclamped, clamped = outputs + _assert_dispatch_matches(unclamped, domain["oracle"]) +> _assert_dispatch_matches(clamped, clamped_oracle) + +tests/moe/test_nvfp4_split_dispatch_policy.py:254: +_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ + +actual = tensor([[ 2.3750, 0.1846, 0.4336, ..., -0.6641, 0.6680, -1.3828], + [-0.0693, -0.1826, -0.1660, ..., 0.226...], + [-0.6133, 0.2158, 0.5938, ..., -0.0352, -0.2012, -0.5859]], + device='cuda:0', dtype=torch.bfloat16) +reference = tensor([[ 1.3594, -0.0732, 0.0640, ..., -0.3535, 0.3438, -1.0234], + [-0.2344, -0.0815, 0.0918, ..., 0.271...], + [-0.5352, 0.0151, 0.3633, ..., -0.1299, -0.0254, -0.5312]], + device='cuda:0', dtype=torch.bfloat16) + + def _assert_dispatch_matches(actual, reference): + assert torch.isfinite(actual).all() + metrics = compare_to_reference(actual, reference) + bound = _bf16_output_bound(reference) +> assert metrics.cos > 0.9999, metrics +E AssertionError: OracleMetrics(max_abs=3.11328125, rmse=0.3658149838447571, mean_abs=0.27279627323150635, cos=0.9297209978103638) +E assert 0.9297209978103638 > 0.9999 +E + where 0.9297209978103638 = OracleMetrics(max_abs=3.11328125, rmse=0.3658149838447571, mean_abs=0.27279627323150635, cos=0.9297209978103638).cos + +tests/moe/test_nvfp4_split_dispatch_policy.py:203: AssertionError +------------------------------ Captured log call ------------------------------- +WARNING b12x.policy.context:context.py:61 b12x policy fallback: moe.decode is using a heuristic on nvidia rtx pro 6000 blackwell workstation edition (compute capability 12.0, 188 SMs) because profile 'nvidia.rtx.pro.6000.blackwell' does not cover the query; query={'activation': 'silu', 'hidden_size': 256, 'intermediate_size': 128, 'num_experts': 8, 'num_tokens': 512, 'quant_mode': 'nvfp4', 'routed_rows': 1024, 'source_format': 'modelopt_nvfp4', 'top_k': 2} +WARNING b12x.policy.context:context.py:61 b12x policy fallback: moe.decode is using a heuristic on nvidia rtx pro 6000 blackwell workstation edition (compute capability 12.0, 188 SMs) because profile 'nvidia.rtx.pro.6000.blackwell' does not cover the query; query={'activation': 'silu', 'hidden_size': 256, 'intermediate_size': 128, 'num_experts': 8, 'num_tokens': 32, 'quant_mode': 'nvfp4', 'routed_rows': 64, 'source_format': 'modelopt_nvfp4', 'top_k': 2} +=========================== short test summary info ============================ +FAILED tests/moe/test_nvfp4_split_dispatch_policy.py::TestNvfp4SplitDispatch::test_swiglu_limit_changes_split_output +1 failed in 13.33s diff --git a/recipes/glm53/evidence/r34-preservation/runtime-metadata.json b/recipes/glm53/evidence/r34-preservation/runtime-metadata.json new file mode 100644 index 00000000..6ad2a1f5 --- /dev/null +++ b/recipes/glm53/evidence/r34-preservation/runtime-metadata.json @@ -0,0 +1,31 @@ +{ + "bytecode_classification": { + "directory_or_other": 1, + "header_only_source_identical": 791, + "source_changed_or_added": 34 + }, + "bytecode_unexplained_differences": [], + "component_untracked_differences": { + "b12x": [ + "opt/glm53-flash/b12x/b12x/attention/mla_compress", + "opt/glm53-flash/b12x/b12x/sequence/_shared/delta_prefill", + "opt/glm53-flash/b12x/b12x/sequence/embedding", + "opt/glm53-flash/b12x/b12x/sequence/engram", + "opt/glm53-flash/b12x/b12x/sequence/gdn_prefill", + "opt/glm53-flash/b12x/benchmarks/moe", + "opt/glm53-flash/b12x/docs/evidence/nvfp4-immutable-input", + "opt/glm53-flash/b12x/validation/deepseek_v41" + ], + "lmcache": [], + "vllm": [ + "opt/glm53-flash/vllm/vllm.egg-info/PKG-INFO", + "opt/glm53-flash/vllm/vllm/_version.py", + "opt/glm53-flash/vllm/vllm/models/deepseek_v4_1", + "opt/glm53-flash/vllm/vllm/models/deepseek_v4_1/common", + "opt/glm53-flash/vllm/vllm/models/deepseek_v4_1/nvidia" + ] + }, + "native_library_differences": [], + "notes": "The base DiffID is identical; this examines all native overrides in the image-specific layer.", + "identical_upper_layer_native_library_count": 670 +} diff --git a/recipes/glm53/evidence/r34-preservation/source-preservation.json b/recipes/glm53/evidence/r34-preservation/source-preservation.json new file mode 100644 index 00000000..a877a4ea --- /dev/null +++ b/recipes/glm53/evidence/r34-preservation/source-preservation.json @@ -0,0 +1,62 @@ +{ + "b12x": { + "automatic_merge_tree": "001237736e737e5889c6d20e6807a17de8e0160f", + "common_ancestor": "6483963275dcf32eb2eec6d100e644d1ea647ed6", + "composition": "98086604c86ec1e78977e5023ce282ecb97ab8a7", + "merge_has_conflicts": true, + "release": "59d51a36a942d56a9c36265855cdc7856fa7712e", + "residual_paths": [ + "M\tb12x/moe/_shared/kernels/dynamic.py", + "M\tb12x/moe/_shared/kernels/nvfp4_phase1.py", + "M\tb12x/moe/_shared/kernels/silu.py", + "M\tb12x/moe/fused_moe/_impl.py", + "M\tb12x/norm/mhc/_impl.py", + "M\tb12x/policy/_profiles/data/nvidia.gb10.48sm.json.gz", + "M\tb12x/policy/_profiles/data/nvidia.rtx.pro.6000.blackwell.json.gz", + "M\tb12x/policy/_profiles/data/nvidia.rtx.pro.6000.blackwell.max-q.json.gz", + "M\tbenchmarks/benchmark_nvfp4_split_materialized.py", + "M\tbenchmarks/gen_nvfp4_split_readme.py", + "D\tbenchmarks/moe/bench_glm_nvfp4_split.py", + "M\tdocs/evidence/nvfp4_split/20260910-rtx5090-nvfp4-split-prefill.json", + "M\tdocs/evidence/nvfp4_split/README.md", + "M\ttests/gemm/test_launch_custom_ops.py", + "M\ttests/gemm/test_mxfp8_linear.py", + "M\ttests/moe/test_cute_migration_moe_standard_corpus.py", + "M\ttests/moe/test_nvfp4_phase_kernels.py", + "M\ttests/moe/test_nvfp4_shared_input_scales.py", + "M\ttests/moe/test_nvfp4_split_backend.py", + "M\ttests/moe/test_nvfp4_split_dispatch_policy.py" + ], + "upstream": "00b69ac22e21413622c4ecd98f607a2c3e015161" + }, + "lmcache": { + "automatic_merge_tree": "5a88a1ea9d2627c76288d056e7193f7454669b64", + "common_ancestor": "7ed4675404a31f4ffafd98975899dc83832ba965", + "composition": "29bc5a2efde737c436b04499eb62cd1776cebeec", + "merge_has_conflicts": false, + "release": "29bc5a2efde737c436b04499eb62cd1776cebeec", + "residual_paths": [], + "upstream": "7ed4675404a31f4ffafd98975899dc83832ba965" + }, + "vllm": { + "automatic_merge_tree": "d48ba1be367143c9ce33d81012d2849ccc0b36a1", + "common_ancestor": "aba94d396d27ea92a2044ad024c003b36da505d3", + "composition": "de982a50c6a3e4718e5cf9f00423a92192718da1", + "merge_has_conflicts": false, + "release": "c496604123b1f4441007b952a7ee37ab12c8f6ad", + "residual_paths": [ + "M\ttests/models/test_glm5next_pooled_indexer.py", + "M\ttests/quantization/test_register_quantization_config.py", + "M\ttests/v1/core/test_boundary_admission.py", + "M\ttests/v1/core/test_boundary_admission_continuation.py", + "M\ttests/v1/core/test_prefill_compute_share_scheduler.py", + "M\ttests/v1/core/test_scheduler.py", + "M\tvllm/distributed/kv_transfer/kv_connector/v1/base.py", + "M\tvllm/models/deepseek_v4_1/quant_config.py", + "M\tvllm/models/glm5next/nvidia/kda.py", + "M\tvllm/v1/attention/backends/mla/sparse_utils.py", + "M\tvllm/v1/core/kv_cache_manager.py" + ], + "upstream": "a6571d0a602ed42525a7a77cf960f39ca7f8d7a8" + } +} diff --git a/recipes/glm53/glm53-r18-lmcache-runtime-requirements.txt b/recipes/glm53/glm53-r18-lmcache-runtime-requirements.txt new file mode 100644 index 00000000..036f34b2 --- /dev/null +++ b/recipes/glm53/glm53-r18-lmcache-runtime-requirements.txt @@ -0,0 +1,30 @@ +# Runtime dependencies added by LMCache to the flattened Jovian Judgement +# serving base. Versions match the dependency set embedded in the qualified +# GLM-5.3-Flash R17 image. The LMCache wheel is installed separately with +# --no-deps so its metadata cannot silently replace Torch, CUDA, or vLLM +# dependencies during a release build. +awscrt==0.36.2 +cufile-python==0.2.0 +google-api-core==2.34.0 +google-auth==2.57.0 +google-cloud-bigtable==2.43.0 +google-cloud-core==2.7.0 +google-cloud-monitoring==2.31.0 +google-crc32c==1.8.0 +grpc-google-iam-v1==0.14.5 +grpcio==1.83.1 +grpcio-status==1.83.1 +importlib_metadata==8.7.1 +opentelemetry-api==1.40.0 +opentelemetry-exporter-otlp==1.40.0 +opentelemetry-exporter-otlp-proto-common==1.40.0 +opentelemetry-exporter-otlp-proto-grpc==1.40.0 +opentelemetry-exporter-otlp-proto-http==1.40.0 +opentelemetry-exporter-prometheus==0.61b0 +opentelemetry-proto==1.40.0 +opentelemetry-sdk==1.40.0 +opentelemetry-semantic-conventions==0.61b0 +prometheus_client==0.24.1 +proto-plus==1.28.4 +pyasn1==0.6.4 +pyasn1_modules==0.4.2 diff --git a/recipes/glm53/glm53_checkpoint_identity.py b/recipes/glm53/glm53_checkpoint_identity.py new file mode 100644 index 00000000..d9933426 --- /dev/null +++ b/recipes/glm53/glm53_checkpoint_identity.py @@ -0,0 +1,172 @@ +#!/usr/bin/env python3 +"""Resolve immutable model identities before enabling persistent checkpoints. + +Repository names are resolved once to a Hugging Face commit; the caller must +pass the returned revision to the model loader. Local checkpoints are identified +by their weight and configuration bytes, not by their directory names. Local +model files must remain unchanged for the lifetime of the serving process. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +from pathlib import Path + + +def local_checkpoint_identity(directory: Path) -> str: + """Hash local safetensors weights and model/tokenizer configuration. + + JSON, Jinja and tokenizer assets are included conservatively. Reading does + not modify the checkpoint. A checkpoint that changes during this operation + is rejected; callers must also keep it immutable during model loading and + serving. Missing weights or unsupported weight formats are errors. + """ + directory = directory.resolve(strict=True) + if not directory.is_dir(): + raise ValueError(f"Checkpoint must be a directory: {directory}") + + def files() -> list[Path]: + return sorted( + path + for path in directory.iterdir() + if path.is_file() + and ( + path.suffix in {".safetensors", ".json", ".jinja", ".model"} + or path.name in {"vocab.txt", "merges.txt", "added_tokens.txt"} + ) + ) + + selected = files() + if not any(path.suffix == ".safetensors" for path in selected): + raise ValueError("Persistent checkpoint identity requires safetensors weights") + if directory / "config.json" not in selected: + raise ValueError("Persistent checkpoint identity requires config.json") + if any(directory.glob("*.bin")) or any(directory.glob("*.gguf")): + raise ValueError("Ambiguous checkpoint: non-safetensors weights are present") + + def content_metadata(stat: os.stat_result) -> tuple[int, ...]: + # Reading may update atime without changing the model. Inode, size, + # mtime and ctime detect replacement or writes without rejecting reads. + return ( + stat.st_dev, + stat.st_ino, + stat.st_size, + stat.st_mtime_ns, + stat.st_ctime_ns, + ) + + before = {path: path.stat() for path in selected} + manifest: list[tuple[str, int, str]] = [] + for path in selected: + with path.open("rb") as stream: + digest = hashlib.file_digest(stream, "sha256").hexdigest() + manifest.append((path.name, before[path].st_size, digest)) + if files() != selected or any( + content_metadata(path.stat()) != content_metadata(before[path]) + for path in selected + ): + raise ValueError("Checkpoint files changed while their identity was computed") + encoded = json.dumps(manifest, separators=(",", ":")).encode() + return hashlib.sha256(b"glm53-local-checkpoint-v1\0" + encoded).hexdigest() + + +def resolve_checkpoint(model: str, revision: str | None = None) -> dict[str, str]: + """Return a content identity and, for a Hub repository, a pinned revision. + + Local files are hashed directly, including symlink targets. For a Hub model, + config.json is resolved through the installed Hugging Face cache client so + its standard authentication, offline and cache settings remain effective. + The returned revision must be used for every weight/configuration download. + """ + path = Path(model).expanduser() + if path.exists(): + return {"identity": local_checkpoint_identity(path), "revision": ""} + if path.is_absolute() or model.startswith(("./", "../", "~")): + raise ValueError(f"Local checkpoint directory does not exist: {model}") + + from huggingface_hub import hf_hub_download + + config_path = Path( + hf_hub_download(model, "config.json", revision=revision or "main") + ) + # The Hub cache client returns snapshots//config.json. Do not + # resolve its symlink: the target blob path no longer contains the commit. + commit = config_path.parent.name + if ( + config_path.parent.parent.name != "snapshots" + or len(commit) != 40 + or any(char not in "0123456789abcdef" for char in commit) + ): + raise ValueError("Hugging Face did not return an immutable snapshot path") + return {"identity": commit, "revision": commit} + + +def resolve_serving_identity( + model: str, + revision: str | None, + draft_model: str | None, + draft_revision: str | None, + speculation: str, + source_lock: Path, +) -> dict[str, object]: + """Identify weights and a build-verified serving source lock. + + ``speculation`` is none, mtp, or dflash. MTP shares target checkpoint weights; + DFlash requires a separate checkpoint. The source lock must describe the + image's installed code and native artifacts; source-code mounts are outside + this launcher's immutable-image contract. + """ + if speculation not in {"none", "mtp", "dflash"}: + raise ValueError("Speculation must be none, mtp, or dflash") + if speculation == "dflash" and not draft_model: + raise ValueError("DFlash requires a draft model") + lock_bytes = source_lock.read_bytes() + if not lock_bytes.startswith(b"format=local-inference-source-lock/v1\n"): + raise ValueError("Unrecognized serving source-lock format") + target = resolve_checkpoint(model, revision) + draft = {"identity": "", "revision": ""} + if speculation == "mtp": + draft = target + elif speculation == "dflash": + assert draft_model is not None + draft = resolve_checkpoint(draft_model, draft_revision) + return { + "checkpoint_identity": { + "target_revision": target["identity"], + "draft_revision": draft["identity"], + "source_revision": hashlib.sha256(lock_bytes).hexdigest(), + }, + "model_revision": target["revision"], + "draft_model_revision": draft["revision"], + } + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--model", required=True) + parser.add_argument("--revision") + parser.add_argument("--draft-model") + parser.add_argument("--draft-revision") + parser.add_argument( + "--speculation", choices=("none", "mtp", "dflash"), required=True + ) + parser.add_argument( + "--source-lock", type=Path, default=Path("/opt/glm53-flash/source.lock") + ) + args = parser.parse_args() + result = resolve_serving_identity( + args.model, + args.revision, + args.draft_model, + args.draft_revision, + args.speculation, + args.source_lock, + ) + print(json.dumps(result, separators=(",", ":"))) + + +if __name__ == "__main__": + main() diff --git a/recipes/glm53/install_glm53_scheduler_overlay.py b/recipes/glm53/install_glm53_scheduler_overlay.py new file mode 100644 index 00000000..490a78ec --- /dev/null +++ b/recipes/glm53/install_glm53_scheduler_overlay.py @@ -0,0 +1,78 @@ +"""Install authenticated Python sources without changing native dependencies.""" + +from __future__ import annotations + +import os +import shutil +import subprocess +from pathlib import Path + +from prepare_glm53_source_bundles import digest, git +from prepare_glm53_scheduler_overlay import INPUTS, PARENT_LOCK_SHA +from source_locked_image_labels import read_lock + + +def main() -> None: + recipe = Path(__file__).resolve().parent + inputs = Path("/inputs") + installed = Path("/opt/glm53-flash/source.lock") + lock = read_lock(inputs / "source.lock") + assert digest(installed) == PARENT_LOCK_SHA + assert digest(inputs / "source.lock") == os.environ["SOURCE_LOCK_SHA256"] + for filename in INPUTS: + assert digest(recipe / filename) == lock[f"input.{filename}.sha256"] + assert digest(inputs / "vllm.bundle") == lock["vllm.bundle.sha256"] + vllm = Path("/opt/glm53-flash/vllm") + flashkda = vllm / "vllm/_flashkda_C.abi3.so" + assert digest(flashkda) == lock["flashkda.extension.sha256"] + assert git(vllm, "rev-parse", "HEAD") == lock["vllm.parent.commit"] + assert not git(vllm, "status", "--porcelain") + subprocess.run( + [ + "bash", + str(recipe / "install_source_bundle.sh"), + str(inputs / "vllm.bundle"), + lock["vllm.commit"], + lock["vllm.tree"], + str(vllm), + ], + check=True, + ) + destinations = { + "serve-glm53-flash-nvfp4-dflash2.sh": "/usr/local/libexec/serve-glm53-flash-nvfp4-dflash2.sh", + "serve-glm53-flash-nvfp4-dflash2-scheduler-qos.sh": "/usr/local/bin/serve-glm53-flash-nvfp4-dflash2.sh", + "serve-glm53-flash-cache-complete.sh": "/usr/local/bin/serve-glm53-flash.sh", + } + for source, destination in destinations.items(): + shutil.copyfile(recipe / source, destination) + Path(destination).chmod(0o755) + # Generated distribution metadata must describe the installed source, not + # the unrelated source revision used to compile unchanged native artifacts. + version = lock["vllm.version"] + (vllm / "vllm/_version.py").write_text( + f"__version__ = version = {version!r}\n" + f"__version_tuple__ = version_tuple = (0, 26, 1, 'rc0', {version.split('+')[1]!r})\n" + ) + metadata = vllm / "vllm.egg-info/PKG-INFO" + lines = metadata.read_text().splitlines(keepends=True) + assert sum(line.startswith("Version: ") for line in lines) == 1 + metadata.write_text( + "".join( + f"Version: {version}\n" if line.startswith("Version: ") else line + for line in lines + ) + ) + shutil.copyfile(inputs / "source.lock", installed) + assert digest(flashkda) == lock["flashkda.extension.sha256"] + for name, source in ( + ("vllm", vllm), + ("b12x", Path("/opt/glm53-flash/b12x")), + ("lmcache", Path("/opt/lmcache/source")), + ): + assert git(source, "rev-parse", "HEAD") == lock[f"{name}.commit"] + assert git(source, "write-tree") == lock[f"{name}.tree"] + assert not git(source, "status", "--porcelain") + + +if __name__ == "__main__": + main() diff --git a/recipes/glm53/install_glm53_source_locked.sh b/recipes/glm53/install_glm53_source_locked.sh new file mode 100644 index 00000000..07ae59c9 --- /dev/null +++ b/recipes/glm53/install_glm53_source_locked.sh @@ -0,0 +1,122 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Build-time mounts: /source-bundles, /build-inputs, /lmcache-artifacts, +# /flashkda-artifacts. No serving source-code mount is required at runtime. +readonly lock=/source-bundles/source.lock +value() { awk -F= -v key="$1" '$1 == key { sub(/^[^=]*=/, ""); print; found=1 } END { if (!found) exit 1 }' "$lock"; } +verify() { test "$(sha256sum "$1" | cut -d' ' -f1)" = "$2"; } +verify "$lock" "${SOURCE_LOCK_SHA256:?Supply the frozen lock hash}" +test "${CACHE_FINGERPRINT:?Supply the frozen cache namespace}" = "$(value runtime.cache.fingerprint)" +verify /source-bundles/uv "$(value build.uv.sha256)" +for name in vllm b12x lmcache; do + verify "/source-bundles/$name.bundle" "$(value "$name.bundle.sha256")" +done +while IFS='=' read -r key checksum; do + if [[ $key == input.*.sha256 ]]; then + filename=${key#input.}; filename=${filename%.sha256} + verify "/build-inputs/$filename" "$checksum" + fi +done < "$lock" +verify /flashkda-artifacts/_flashkda_C.abi3.so "$(value flashkda.extension.sha256)" +verify /flashkda-artifacts/_C_stable_libtorch.abi3.so "$(value vllm.native.extension.sha256)" +verify /flashkda-artifacts/native-source.identity "$(value vllm.native.identity.sha256)" +verify /source-bundles/lmcache-mp-wrapper.sh "$(value launcher.lmcache.sha256)" +verify /flashinfer-wheels/flashinfer_python-0.6.18+cu133-py3-none-any.whl "$(value flashinfer.python.wheel.sha256)" +verify /flashinfer-wheels/flashinfer_jit_cache-0.6.18+cu133-cp39-abi3-manylinux_2_28_x86_64.whl "$(value flashinfer.jit-cache.wheel.sha256)" +test "$(cat /flashkda-artifacts/flashkda-base.commit)" = "$(value flashkda.base.commit)" +test "$(awk '{print $1}' /flashkda-artifacts/flashkda-patch.sha256)" = "$(value flashkda.patch.sha256)" + +for name in vllm b12x lmcache; do + if [[ $name == lmcache ]]; then destination=/opt/lmcache/source; else destination=/opt/glm53-flash/$name; fi + bash /build-inputs/install_source_bundle.sh "/source-bundles/$name.bundle" \ + "$(value "$name.commit")" "$(value "$name.tree")" "$destination" +done +verify /opt/glm53-flash/vllm/cmake/external_projects/patches/flashkda-packed-checkpoints.patch \ + "$(value flashkda.patch.sha256)" + +readonly python=/opt/venv/bin/python +readonly uv=/source-bundles/uv +readonly lmcache_package=/opt/venv/lib/python3.12/site-packages/lmcache +"$uv" run --no-project --python "$python" "$python" \ + /build-inputs/install_vllm_source_version.py /opt/glm53-flash/vllm \ + "$(value vllm.version)" +torch_before=$("$uv" run --no-project --python "$python" "$python" -c \ + 'import torch; print(f"{torch.__version__}|{torch.version.cuda}|{int(torch._C._GLIBCXX_USE_CXX11_ABI)}")') +test "$torch_before" = '2.13.0|13.3|1' +"$uv" pip install --python "$python" --no-deps --requirement \ + /build-inputs/glm53-r18-lmcache-runtime-requirements.txt +"$uv" pip install --python "$python" --no-deps --reinstall /lmcache-artifacts/*.whl +"$uv" pip install --python "$python" --no-deps --reinstall /flashinfer-wheels/*.whl +# Wheels omit some package data. Install the complete committed package, while +# retaining the native extensions compiled in the dedicated build stage. +cp -a /opt/lmcache/source/lmcache/. "$lmcache_package/" +GIT_WORK_TREE=/opt/venv/lib/python3.12/site-packages \ + git --git-dir=/opt/lmcache/source/.git diff --quiet HEAD -- lmcache +install -Dm755 /flashkda-artifacts/_flashkda_C.abi3.so \ + /opt/glm53-flash/vllm/vllm/_flashkda_C.abi3.so +install -Dm755 /flashkda-artifacts/_flashkda_C.abi3.so \ + /opt/venv/lib/python3.12/site-packages/vllm/_flashkda_C.abi3.so +install -Dm755 /lmcache-artifacts/liblmcache_cumem_shareable.so \ + /opt/lmcache/lib/liblmcache_cumem_shareable.so +install -Dm644 /lmcache-artifacts/SHA256SUMS /opt/lmcache/native-artifacts.sha256 +for directory in /opt/glm53-flash/vllm/vllm /opt/venv/lib/python3.12/site-packages/vllm; do + install -Dm755 /flashkda-artifacts/_C_stable_libtorch.abi3.so \ + "$directory/_C_stable_libtorch.abi3.so" +done +install -Dm644 /flashkda-artifacts/native-source.identity /opt/glm53-flash/native-source.identity +# One import root per library: callers that clear PYTHONPATH must not reach +# Python packages inherited from the runtime foundation. Canonical source +# directories already retain every required native artifact and package data. +test -d /opt/venv/lib/python3.12/site-packages/vllm +test -d /opt/venv/lib/python3.12/site-packages/b12x +test ! -L /opt/venv/lib/python3.12/site-packages/vllm +test ! -L /opt/venv/lib/python3.12/site-packages/b12x +rm -r -- /opt/venv/lib/python3.12/site-packages/vllm /opt/venv/lib/python3.12/site-packages/b12x +ln -s /opt/glm53-flash/vllm/vllm /opt/venv/lib/python3.12/site-packages/vllm +ln -s /opt/glm53-flash/b12x/b12x /opt/venv/lib/python3.12/site-packages/b12x +install -Dm755 /source-bundles/lmcache-mp-wrapper.sh /usr/local/bin/lmcache-mp-wrapper.sh +install -Dm755 /build-inputs/serve-ds4-jovian.sh /usr/local/bin/serve-ds4-jovian.sh + +install -Dm755 /build-inputs/serve-glm53-flash-nvfp4-dflash2.sh /usr/local/libexec/serve-glm53-flash-nvfp4-dflash2.sh +install -Dm755 /build-inputs/serve-glm53-flash-nvfp4-dflash2-scheduler-qos.sh /usr/local/bin/serve-glm53-flash-nvfp4-dflash2.sh +install -Dm755 /build-inputs/serve-glm53-flash-lmcache.sh /usr/local/bin/serve-glm53-flash-lmcache.sh +install -Dm755 /build-inputs/serve-glm53-flash-lmcache-cache-complete.sh /usr/local/libexec/serve-glm53-flash-lmcache-cache-complete.sh +install -Dm755 /build-inputs/serve-glm53-flash-cache-complete.sh /usr/local/bin/serve-glm53-flash.sh +install -Dm755 /build-inputs/glm53_checkpoint_identity.py /usr/local/libexec/glm53_checkpoint_identity.py +install -Dm644 "$lock" /opt/glm53-flash/source.lock +# BuildKit has no NVIDIA driver mount. The stable extension links the driver +# API, so schema/import inspection uses the toolkit stub only for this process. +# Runtime imports and kernels are validated separately with the real driver. +mkdir -p /tmp/native-import-driver-stub +ln -s /usr/local/cuda/lib64/stubs/libcuda.so /tmp/native-import-driver-stub/libcuda.so.1 +LD_LIBRARY_PATH="/tmp/native-import-driver-stub:${LD_LIBRARY_PATH:-}" \ +PYTHONPATH= \ +CUDA_VISIBLE_DEVICES= "$uv" run --no-project --python "$python" "$python" -c ' +import importlib.metadata as metadata +from pathlib import Path +import torch, vllm, b12x, vllm._flashkda_C +import vllm._C_stable_libtorch +import vllm.vllm_flash_attn.layers.rotary, vllm.third_party.triton_kernels.topk +import lmcache.cuda_ops, lmcache.lmcache_native, lmcache.lmcache_fs +from lmcache.integration.vllm.recurrent_checkpoint_connector import LMCacheRecurrentCheckpointConnector +assert (torch.__version__, torch.version.cuda, torch._C._GLIBCXX_USE_CXX11_ABI) == ("2.13.0", "13.3", True) +print("LMCache wheel:", metadata.version("lmcache")) +print("Source imports:", vllm.__file__, b12x.__file__) +assert Path(vllm.__file__).resolve() == Path("/opt/glm53-flash/vllm/vllm/__init__.py") +assert Path(b12x.__file__).resolve() == Path("/opt/glm53-flash/b12x/b12x/__init__.py") +schema = str(torch.ops._C.fused_deepseek_v4_qnorm_rope_kv_rope_quant_insert.default._schema) +assert "q_out" in schema and "q_head_padded" not in schema, schema +assert schema.endswith("-> ()"), schema +' +unlink /tmp/native-import-driver-stub/libcuda.so.1 +rmdir /tmp/native-import-driver-stub +# The foundation exposes both venv and system site-packages. Check the exact +# interpreter search path used by serving, including the system Torch package. +"$uv" run --no-project --python "$python" "$python" -m pip check +for name in vllm b12x lmcache; do + if [[ $name == lmcache ]]; then destination=/opt/lmcache/source; else destination=/opt/glm53-flash/$name; fi + test "$(git -C "$destination" rev-parse HEAD)" = "$(value "$name.commit")" + test "$(git -C "$destination" rev-parse 'HEAD^{tree}')" = "$(value "$name.tree")" + git -C "$destination" diff --quiet HEAD +done diff --git a/recipes/glm53/install_source_bundle.sh b/recipes/glm53/install_source_bundle.sh new file mode 100644 index 00000000..b8be1c4f --- /dev/null +++ b/recipes/glm53/install_source_bundle.sh @@ -0,0 +1,55 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Install a complete committed source tree while retaining ignored runtime +# artifacts, such as compiled extensions and separately packaged dependencies. +# The caller supplies an exact commit and tree from the serving source lock. +if (($# != 4)); then + echo "usage: $0 " >&2 + exit 2 +fi +readonly bundle=$(realpath -e "$1") +readonly commit=$2 tree=$3 destination=$(realpath -m "$4") +if [[ ! $commit =~ ^[0-9a-f]{40}$ || ! $tree =~ ^[0-9a-f]{40}$ ]]; then + echo "Source installation requires full Git commit and tree identifiers." >&2 + exit 2 +fi +if [[ $destination == / || $destination == /root || $destination == /opt ]]; then + echo "The destination must be a dedicated source repository." >&2 + exit 2 +fi +if ! git bundle list-heads "$bundle" | awk -v commit="$commit" ' + $1 == commit { found = 1 } END { exit !found }'; then + echo "The bundle does not advertise the locked source commit." >&2 + exit 1 +fi +if [[ ! -d $destination/.git ]]; then + if [[ -e $destination && -n $(ls -A "$destination") ]]; then + echo "Refusing to initialize a nonempty non-repository destination." >&2 + exit 1 + fi + git init --quiet "$destination" +fi +if [[ $(git -C "$destination" rev-parse --show-toplevel) != "$destination" ]]; then + echo "The destination is not its own Git working tree." >&2 + exit 1 +fi +git -C "$destination" fetch --quiet --no-tags "$bundle" "$commit" +if [[ $(git -C "$destination" rev-parse "${commit}^{tree}") != "$tree" ]]; then + echo "The bundle's source tree differs from the source lock." >&2 + exit 1 +fi + +# Deletions are limited to paths tracked by the image's existing index and +# absent from the locked tree. Git objects retain their original contents. +readonly previous_tree=$(git -C "$destination" write-tree) +while IFS= read -r -d '' path; do + rm -f -- "$destination/$path" +done < <(git -C "$destination" diff --name-only --diff-filter=D -z \ + "$previous_tree" "$commit") +git -C "$destination" update-ref refs/heads/image-source "$commit" +git -C "$destination" symbolic-ref HEAD refs/heads/image-source +git -C "$destination" read-tree "$commit" +git -C "$destination" checkout-index --all --force +git -C "$destination" diff --quiet HEAD +test "$(git -C "$destination" write-tree)" = "$tree" diff --git a/recipes/glm53/install_vllm_source_version.py b/recipes/glm53/install_vllm_source_version.py new file mode 100644 index 00000000..ad3fae2f --- /dev/null +++ b/recipes/glm53/install_vllm_source_version.py @@ -0,0 +1,38 @@ +"""Identify installed Python sources independently of native build provenance.""" + +from __future__ import annotations + +import argparse +from pathlib import Path + +from packaging.version import Version + + +def install_version(source: Path, version: str) -> None: + parsed = Version(version) + if parsed.local is None: + raise ValueError("Serving version must include a source identity") + metadata = source / "vllm.egg-info/PKG-INFO" + lines = metadata.read_text().splitlines(keepends=True) + if sum(line.startswith("Version: ") for line in lines) != 1: + raise ValueError("Distribution metadata must have exactly one Version field") + prerelease = (f"{parsed.pre[0]}{parsed.pre[1]}",) if parsed.pre else () + version_tuple = (*parsed.release, *prerelease, parsed.local) + (source / "vllm/_version.py").write_text( + f"__version__ = version = {version!r}\n" + f"__version_tuple__ = version_tuple = {version_tuple!r}\n" + ) + metadata.write_text( + "".join( + f"Version: {version}\n" if line.startswith("Version: ") else line + for line in lines + ) + ) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("source", type=Path) + parser.add_argument("version") + args = parser.parse_args() + install_version(args.source, args.version) diff --git a/recipes/glm53/pack_lmcache_native_reuse.py b/recipes/glm53/pack_lmcache_native_reuse.py new file mode 100644 index 00000000..dd14f520 --- /dev/null +++ b/recipes/glm53/pack_lmcache_native_reuse.py @@ -0,0 +1,178 @@ +"""Package LMCache wheels with source- and ABI-matched compiled extensions. + +The reference image must contain the same native sources and build policy. +Compiled payloads are copied byte-for-byte; wheel tags and RECORD describe the +resulting platform wheel. Native-source changes fail rather than reusing an +incompatible binary. + +``cpu-rebuild-cuda-reuse`` requires a wheel containing rebuilt common C++ +extensions and preserves only the CUDA extension. Every native input outside +the two CPU implementation directories must still match the reference. +""" + +from __future__ import annotations + +import argparse +import base64 +import csv +import hashlib +import io +from pathlib import Path +import subprocess +import zipfile + +NATIVE_INPUTS = ( + "csrc", + "rust", + "setup_extensions", + "CMakeLists.txt", + "setup.py", + "pyproject.toml", + "MANIFEST.in", + "requirements", +) + + +CPU_SOURCE_DIRS = ("csrc/lmcache_native/", "csrc/storage_backends/") +CPU_EXTENSIONS = {"lmcache_native", "lmcache_fs", "lmcache_redis"} + + +def native_identity(root: Path, mode: str = "reuse-all") -> dict[str, str]: + identity = { + name: subprocess.check_output( + ["git", "-C", str(root), "rev-parse", f"HEAD:{name}"], text=True + ).strip() + for name in NATIVE_INPUTS + if mode == "reuse-all" or name != "csrc" + } + if mode == "cpu-rebuild-cuda-reuse": + sources = subprocess.check_output( + ["git", "-C", str(root), "ls-tree", "-r", "HEAD", "csrc"], text=True + ) + for line in sources.splitlines(): + metadata, path = line.split("\t", 1) + if not path.startswith(CPU_SOURCE_DIRS): + identity[path] = metadata + return identity + + +def select_native( + files: dict[str, bytes], native: dict[str, bytes], mode: str +) -> dict[str, bytes]: + if mode == "reuse-all": + if any(path.endswith(".so") for path in files): + raise ValueError("Full native reuse requires a Python-only wheel") + return native + compiled = {path for path in files if path.endswith(".so")} + if {Path(path).name.split(".")[0] for path in compiled} != CPU_EXTENSIONS: + raise ValueError("CUDA-only reuse requires all three rebuilt CPU extensions") + if any(not path.startswith("lmcache/") for path in compiled): + raise ValueError("Rebuilt CPU extensions must belong to lmcache") + cuda = { + path: data + for path, data in native.items() + if Path(path).name.startswith("cuda_ops.") + } + if len(cuda) != 1: + raise ValueError("Expected exactly one reference CUDA extension") + return cuda + + +def platform_wheel( + files: dict[str, bytes], native: dict[str, bytes], tag: str +) -> dict[str, bytes]: + """Return complete wheel members with consistent platform metadata and hashes.""" + if not native or any( + not path.startswith("lmcache/") or not path.endswith(".so") for path in native + ): + raise ValueError("Expected compiled extensions in the lmcache package") + wheel_paths = [name for name in files if name.endswith(".dist-info/WHEEL")] + if len(wheel_paths) != 1 or tag != "cp312-cp312-linux_x86_64": + raise ValueError("Expected one CPython 3.12 Linux x86_64 wheel") + result = {**files, **native} + wheel_path = wheel_paths[0] + lines = [ + line + for line in files[wheel_path].decode().splitlines() + if line and not line.startswith(("Tag:", "Root-Is-Purelib:")) + ] + result[wheel_path] = ( + "\n".join(lines + ["Root-Is-Purelib: false", f"Tag: {tag}"]) + "\n" + ).encode() + record = wheel_path.removesuffix("WHEEL") + "RECORD" + rows = io.StringIO(newline="") + writer = csv.writer(rows, lineterminator="\n") + for name, data in sorted(result.items()): + if name != record: + digest = ( + base64.urlsafe_b64encode(hashlib.sha256(data).digest()) + .rstrip(b"=") + .decode() + ) + writer.writerow([name, f"sha256={digest}", len(data)]) + writer.writerow([record, "", ""]) + result[record] = rows.getvalue().encode() + return result + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--source", type=Path, required=True) + parser.add_argument("--reference-source", type=Path, required=True) + parser.add_argument("--reference-package", type=Path, required=True) + parser.add_argument("--wheel-dir", type=Path, required=True) + parser.add_argument( + "--mode", choices=("reuse-all", "cpu-rebuild-cuda-reuse"), default="reuse-all" + ) + args = parser.parse_args() + if native_identity(args.source, args.mode) != native_identity( + args.reference_source, args.mode + ): + raise ValueError( + "LMCache native build inputs differ; provide compatible compiled artifacts" + ) + wheels = list(args.wheel_dir.glob("*.whl")) + suffix = ( + "-py3-none-any.whl" + if args.mode == "reuse-all" + else "-cp312-cp312-linux_x86_64.whl" + ) + if len(wheels) != 1 or not wheels[0].name.endswith(suffix): + raise ValueError(f"Expected one LMCache wheel ending in {suffix}") + source = wheels[0] + reference_metadata = list( + args.reference_package.parent.glob("lmcache-*.dist-info/WHEEL") + ) + if len(reference_metadata) != 1: + raise ValueError("Expected one installed LMCache distribution") + tags = [ + line.removeprefix("Tag: ") + for line in reference_metadata[0].read_text().splitlines() + if line.startswith("Tag: ") + ] + if tags != ["cp312-cp312-linux_x86_64"]: + raise ValueError(f"Unsupported reference wheel tags: {tags}") + native = { + "lmcache/" + str(path.relative_to(args.reference_package)): path.read_bytes() + for path in args.reference_package.rglob("*.so") + } + with zipfile.ZipFile(source) as archive: + files = {name: archive.read(name) for name in archive.namelist()} + native = select_native(files, native, args.mode) + output = source.with_name(source.name.removesuffix(suffix) + f"-{tags[0]}.whl") + if output != source and output.exists(): + raise FileExistsError(output) + staged = output.with_suffix(".whl.tmp") + if staged.exists(): + raise FileExistsError(staged) + with zipfile.ZipFile(staged, "w", compression=zipfile.ZIP_DEFLATED) as archive: + for name, data in platform_wheel(files, native, tags[0]).items(): + archive.writestr(name, data) + staged.replace(output) + if source != output: + source.unlink() + print(f"Preserved {len(native)} compiled LMCache artifacts in {output.name}") + + +if __name__ == "__main__": + main() diff --git a/recipes/glm53/prepare_glm53_scheduler_overlay.py b/recipes/glm53/prepare_glm53_scheduler_overlay.py new file mode 100644 index 00000000..5850aa73 --- /dev/null +++ b/recipes/glm53/prepare_glm53_scheduler_overlay.py @@ -0,0 +1,110 @@ +"""Freeze a Python-only scheduler overlay over the immutable FP8 serving image.""" + +from __future__ import annotations + +import argparse +import shutil +import subprocess +from pathlib import Path + +from prepare_glm53_source_bundles import digest, git +from source_locked_image_labels import read_lock + +PARENT = "voipmonitor/vllm@sha256:f5f121e37fd2afbb6f8f036e7eb627435cfb736de0a4420306dc2a25b6631669" +PARENT_LOCK_SHA = "15a9a649559830822cb943ea0c3c6a644c8b69c9e54d7be3e265801851843932" +INPUTS = ( + "Dockerfile.glm53-scheduler-overlay", + "build_glm53_scheduler_overlay.sh", + "prepare_glm53_scheduler_overlay.py", + "prepare_glm53_source_bundles.py", + "install_glm53_scheduler_overlay.py", + "install_source_bundle.sh", + "source_locked_image_labels.py", + "serve-glm53-flash-nvfp4-dflash2.sh", + "serve-glm53-flash-nvfp4-dflash2-scheduler-qos.sh", + "serve-glm53-flash-cache-complete.sh", +) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + for name in ("vllm", "parent-lock", "uv", "output"): + parser.add_argument(f"--{name}", type=Path, required=True) + parser.add_argument("--release-name", required=True) + parser.add_argument("--release-version", required=True) + args = parser.parse_args() + if digest(args.parent_lock) != PARENT_LOCK_SHA: + raise ValueError("The overlay requires the authenticated R28 source lock") + root = args.vllm.resolve() + if git(root, "status", "--porcelain"): + raise ValueError("vLLM must be a clean committed source tree") + lock = read_lock(args.parent_lock) + parent_commit = lock["vllm.commit"] + git(root, "merge-base", "--is-ancestor", parent_commit, "HEAD") + changed = git(root, "diff", "--name-only", parent_commit, "HEAD").splitlines() + allowed = { + "vllm/config/scheduler.py", + "vllm/v1/core/sched/prefill_interleave.py", + "vllm/v1/core/sched/scheduler.py", + "vllm/v1/engine/core.py", + "tests/v1/core/test_prefill_compute_share_scheduler.py", + "tests/v1/core/test_scheduler.py", + "tests/v1/engine/test_compute_fairness_feedback.py", + } + if not changed or set(changed) - allowed: + raise ValueError(f"Changes exceed the Python scheduler contract: {changed}") + args.output = args.output.resolve() + args.output.mkdir(parents=True, exist_ok=False) + bundle = args.output / "vllm.bundle" + subprocess.run( + [ + "git", + "-C", + str(root), + "bundle", + "create", + str(bundle), + "HEAD", + f"^{parent_commit}", + ], + check=True, + ) + lock.update( + { + "release.name": args.release_name, + "release.version": args.release_version, + "runtime.parent.release-image": PARENT, + "runtime.parent.source-lock.sha256": PARENT_LOCK_SHA, + "runtime.rootfs.layers": "3", + "runtime.default.reasoning-effort": "high", + "runtime.scheduler.max-parallel-prefills": "1", + "runtime.scheduler.auto-max-parallel-prefills": "4", + "vllm.parent.commit": parent_commit, + "vllm.commit": git(root, "rev-parse", "HEAD"), + "vllm.tree": git(root, "rev-parse", "HEAD^{tree}"), + "vllm.package.tree": git(root, "rev-parse", "HEAD:vllm"), + "vllm.bundle.sha256": digest(bundle), + "vllm.bundle.mode": "incremental over parent image; complete installed Git history", + } + ) + lock["vllm.version"] = ( + f"0.26.1rc0+glm53.{args.release_version}.vllm{lock['vllm.commit'][:8]}" + ) + recipe = Path(__file__).resolve().parent + for filename in INPUTS: + lock[f"input.{filename}.sha256"] = digest(recipe / filename) + shutil.copy2(args.uv, args.output / "uv") + lock["build.uv.sha256"] = digest(args.output / "uv") + lock["runtime.cache.fingerprint"] = ( + f"cu133-torch213-glm53-vllm{lock['vllm.package.tree'][:8]}" + f"-b12x{lock['b12x.package.tree'][:8]}-lmcache{lock['lmcache.package.tree'][:8]}" + f"-flashkda{lock['flashkda.extension.sha256'][:8]}" + ) + (args.output / "source.lock").write_text( + "".join(f"{key}={value}\n" for key, value in lock.items()) + ) + print(f"SOURCE_LOCK_SHA256={digest(args.output / 'source.lock')}") + + +if __name__ == "__main__": + main() diff --git a/recipes/glm53/prepare_glm53_source_bundles.py b/recipes/glm53/prepare_glm53_source_bundles.py new file mode 100644 index 00000000..34affe20 --- /dev/null +++ b/recipes/glm53/prepare_glm53_source_bundles.py @@ -0,0 +1,200 @@ +"""Freeze committed serving sources and build inputs for a flat Docker image.""" + +from __future__ import annotations + +import argparse +import hashlib +import shutil +import subprocess +from pathlib import Path +from urllib.parse import urlsplit + + +def git(root: Path, *args: str) -> str: + return subprocess.check_output(["git", "-C", str(root), *args], text=True).strip() + + +def digest(path: Path) -> str: + with path.open("rb") as stream: + return hashlib.file_digest(stream, "sha256").hexdigest() + + +def repository_url(value: str) -> str: + if value.startswith("git@github.com:"): + value = "https://github.com/" + value.removeprefix("git@github.com:") + parsed = urlsplit(value) + if ( + parsed.scheme != "https" + or not parsed.hostname + or not parsed.path.strip("/") + or parsed.username is not None + or parsed.password is not None + or parsed.query + or parsed.fragment + or any(c.isspace() for c in value) + ): + raise ValueError("Source repository must be an HTTPS URL without credentials") + return value.rstrip("/") + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + for name in ("vllm", "b12x", "lmcache", "native-artifact", "uv", "output"): + parser.add_argument(f"--{name}", type=Path, required=True) + for name in ("vllm", "b12x", "lmcache"): + parser.add_argument( + f"--{name}-repository", + help="Published source repository; defaults to origin", + ) + parser.add_argument( + "--release-name", default="jovian-judgement-community-source-locked" + ) + parser.add_argument("--release-version", default="source-locked") + parser.add_argument( + "--lmcache-native-mode", + choices=("reuse-all", "cpu-rebuild-cuda-reuse"), + default="reuse-all", + ) + args = parser.parse_args() + args.output = args.output.resolve() + roots = { + name: getattr(args, name).resolve() for name in ("vllm", "b12x", "lmcache") + } + for name, root in roots.items(): + if git(root, "status", "--porcelain"): + raise ValueError(f"{name} must have a clean committed source tree") + args.output.mkdir(parents=True, exist_ok=False) + lock = { + "format": "local-inference-source-lock/v1", + "implementation.status": "implemented", + "qualification.contract": "See evidence for the resulting immutable image", + "release.name": args.release_name, + "release.version": args.release_version, + "runtime.base.image": "voipmonitor/vllm@sha256:93ac5228f1cbde2182ca294d8b479259144742af2756a49ff207dd245429bf43", + "runtime.cuda.version": "13.3", + "runtime.pytorch.version": "2.13.0", + "runtime.rootfs.layers": "2", + "runtime.cudagraph.mode": "FULL_AND_PIECEWISE", + "runtime.moe-backend.default": "b12x", + "runtime.scheduler.max-num-batched-tokens": "4096", + "runtime.scheduler.prefill-compute-share": "0.4", + "runtime.nccl.channels": "16", + "runtime.nccl.buffer.bytes": "2097152", + "runtime.recurrent-checkpoint-policy": "auto", + "lmcache.native.artifact.image": "localinferencelab/vllm@sha256:3ae04d964f8e7e936ef4b34dd75406b9169fead8062fb8b4b1634f3bbf512827", + "lmcache.native.mode": args.lmcache_native_mode, + "runtime.lmcache.transfer": "engine-driven asynchronous shared memory", + "runtime.lmcache.checkpoints": "atomic target and draft bundles for request_boundaries; independent chunks for explicit aligned", + "model.repository": "local-inference-lab/GLM-5.3-Flash-NVFP4", + "draft.repository": "local-inference-lab/GLM-5.3-Flash-DFlash2", + "draft.quantization": "MXFP8", + } + for name, root in roots.items(): + revision = git(root, "rev-parse", "HEAD") + lock[f"{name}.repository"] = repository_url( + getattr(args, f"{name}_repository") + or git(root, "remote", "get-url", "origin") + ) + lock[f"{name}.commit"] = revision + lock[f"{name}.tree"] = git(root, "rev-parse", "HEAD^{tree}") + lock[f"{name}.package.tree"] = git(root, "rev-parse", f"HEAD:{name}") + bundle = args.output / f"{name}.bundle" + subprocess.run( + ["git", "-C", str(root), "bundle", "create", str(bundle), "HEAD"], + check=True, + ) + lock[f"{name}.bundle.sha256"] = digest(bundle) + lock["lmcache.version"] = ( + f"0.5.5.dev0+glm53checkpoints.{lock['lmcache.commit'][:8]}" + ) + lock["vllm.version"] = ( + f"0.26.1rc0+glm53.{args.release_version}.vllm{lock['vllm.commit'][:8]}" + ) + patch = "cmake/external_projects/patches/flashkda-packed-checkpoints.patch" + lock["flashkda.base.commit"] = ( + (args.native_artifact / "flashkda-base.commit").read_text().strip() + ) + lock["flashkda.patch.sha256"] = digest(roots["vllm"] / patch) + recorded_patch = ( + (args.native_artifact / "flashkda-patch.sha256").read_text().split()[0] + ) + if recorded_patch != lock["flashkda.patch.sha256"]: + raise ValueError( + "Native artifact patch differs from the committed vLLM dependency" + ) + lock["flashkda.extension.sha256"] = digest( + args.native_artifact / "_flashkda_C.abi3.so" + ) + native_identity = dict( + line.split("=", 1) + for line in (args.native_artifact / "native-source.identity") + .read_text() + .splitlines() + ) + for field, path in ( + ("csrc.tree", "csrc"), + ("cmake.tree", "cmake"), + ("cmakelists.blob", "CMakeLists.txt"), + ): + expected = git(roots["vllm"], "rev-parse", f"HEAD:{path}") + if native_identity[f"vllm.{field}"] != expected: + raise ValueError(f"Stable native artifact has mismatched {path} sources") + lock[f"vllm.native.{field}"] = expected + lock["vllm.native.extension.sha256"] = digest( + args.native_artifact / "_C_stable_libtorch.abi3.so" + ) + lock["vllm.native.identity.sha256"] = digest( + args.native_artifact / "native-source.identity" + ) + lock["flashinfer.commit"] = "803c4664f4771ddc418f20a57f752469a237a825" + lock["flashinfer.artifact.image"] = ( + "voipmonitor/vllm@sha256:79edbc91874d9468e3e6268e1584503e3dec55f2a4d3bdd70d5c43e9b41675c7" + ) + lock["flashinfer.python.wheel.sha256"] = ( + "3a5b1143d8b933d71881c2d71a2c0677360528344813be1893bafa9f335affea" + ) + lock["flashinfer.jit-cache.wheel.sha256"] = ( + "9d08679e4ba3cc7c49c5f8035268b7958e9c4d8c7a5e8cd842211da46007809f" + ) + docker = Path(__file__).resolve().parent + lmcache_launcher = docker.parents[1] / "launchers/lmcache-mp-wrapper.sh" + shutil.copy2(lmcache_launcher, args.output / "lmcache-mp-wrapper.sh") + lock["launcher.lmcache.sha256"] = digest(lmcache_launcher) + inputs = ( + "Dockerfile.glm53-cache-contracts", + "build_glm53_cache_contract_image.sh", + "source_locked_image_labels.py", + "install_source_bundle.sh", + "pack_lmcache_native_reuse.py", + "install_glm53_source_locked.sh", + "install_vllm_source_version.py", + "serve-ds4-jovian.sh", + "Dockerfile.jovian-stable-native", + "build_jovian_stable_native.py", + "serve-glm53-flash-nvfp4-dflash2.sh", + "serve-glm53-flash-nvfp4-dflash2-scheduler-qos.sh", + "serve-glm53-flash-lmcache.sh", + "serve-glm53-flash-lmcache-cache-complete.sh", + "serve-glm53-flash-cache-complete.sh", + "glm53_checkpoint_identity.py", + "glm53-r18-lmcache-runtime-requirements.txt", + ) + for filename in inputs: + lock[f"input.{filename}.sha256"] = digest(docker / filename) + shutil.copy2(args.uv, args.output / "uv") + lock["build.uv.sha256"] = digest(args.output / "uv") + lock["runtime.cache.fingerprint"] = ( + f"cu133-torch213-glm53-vllm{lock['vllm.package.tree'][:8]}" + f"-b12x{lock['b12x.package.tree'][:8]}-lmcache{lock['lmcache.package.tree'][:8]}" + f"-flashkda{lock['flashkda.extension.sha256'][:8]}" + f"-native{lock['vllm.native.extension.sha256'][:8]}" + f"-fi{lock['flashinfer.commit'][:8]}" + ) + target = args.output / "source.lock" + target.write_text("".join(f"{key}={value}\n" for key, value in lock.items())) + print(f"SOURCE_LOCK_SHA256={digest(target)}") + print(f"CACHE_FINGERPRINT={lock['runtime.cache.fingerprint']}") + + +if __name__ == "__main__": + main() diff --git a/recipes/glm53/r34-preservation.md b/recipes/glm53/r34-preservation.md new file mode 100644 index 00000000..46a441dd --- /dev/null +++ b/recipes/glm53/r34-preservation.md @@ -0,0 +1,153 @@ +# R34 runtime preservation and NVFP4 activation correctness + +Status: **qualified for source/artifact preservation and the focused tests below**. +Audit date: 2026-09-11. +The audit compares the published R34 image with the public PR composition in +[vLLM issue #731](https://github.com/local-inference-lab/vllm/issues/731). +It does not claim a complete serving qualification for every model or topology. + +Recorded evidence: [installed-file checks](evidence/r34-preservation/artifact-summary.json), +[independent merge residuals](evidence/r34-preservation/source-preservation.json), +[decompressed profile checks](evidence/r34-preservation/profile-preservation.json), +and [runtime metadata classification](evidence/r34-preservation/runtime-metadata.json). + +## Conclusion + +No required R34 serving implementation was found missing from the composed +image. The composition retains the R34 launchers, cache implementation, +speculation paths and retained B12X tuning, with the pinned JJ/master changes. +It is **not** the same source tree as R34 and must not be described as such. + +One intentional mathematical difference is essential: the composition fixes +the missing SwiGLU clamp in R34's split NVFP4 MoE kernel. Preserving R34's +serving functionality does not mean preserving that defect. The correction is +already merged in [B12X #353](https://github.com/local-inference-lab/b12x/pull/353) +and is present in the pinned B12X base; no additional fix PR is required. + +## Immutable comparison boundary + +| Artifact | Identity | +|---|---| +| Published R34 | `localinferencelab/vllm@sha256:d2d13141fc158f3e5f989930c4be4637eaf28322307288724c2faa7cd4e9bcc7` | +| Composed image, local only | image ID `sha256:d3508d5afd3c616db55efb5d18490e12370fdd0f9c6821f434136e4f4fc21c6a` | +| R34 source-lock SHA-256 | `e7b5712d12676c8daf0a000398cfa2d57eedb3e290cedf611fcee28fe2413dd0` | +| Composition source-lock SHA-256 | `7be25b839e3c34b68e5f54dbec6d12b2cb540277966e49c424e082e57c448e88` | + +| Component | R34 commit | Composed commit | Installed tracked files, R34 / composition | +|---|---|---|---:| +| vLLM | `c496604123b1f4441007b952a7ee37ab12c8f6ad` | `de982a50c6a3e4718e5cf9f00423a92192718da1` | 6,835 / 6,870 | +| B12X | `59d51a36a942d56a9c36265855cdc7856fa7712e` | `98086604c86ec1e78977e5023ce282ecb97ab8a7` | 881 / 1,015 | +| LMCache | `29bc5a2efde737c436b04499eb62cd1776cebeec` | same | 2,423 / 2,423 | + +Every tracked file and executable mode matches its declared Git tree in both +images: **20,447 checks, zero mismatches**. Both source-lock hashes match the +OCI labels. This checks actual installed files, not just the labels. + +## Independent source-preservation check + +For each repository, `git merge-tree --write-tree R34_COMMIT PINNED_BASE` +combines the released source and the base from the public review manifest. +The result is compared with the composed commit. This is independent of +replaying the PR manifest against its own expected tree. + +| Area | Residual review and result | +|---|---| +| vLLM | Eleven residual paths: native-import isolation from #730; equivalent connector-method and field ordering; documentation/formatting; test maintenance and additional pending-import lane coverage. No R34 runtime implementation is removed. | +| B12X split MoE | Retains route/compute splitting, fast preparation, low-SMEM controls and numerical-recipe propagation. Uses the merged clamp and scratch-sizing helper; removes a duplicated allocation branch rather than removing allocation. | +| B12X MHC | Retains capacity-based partial grouping while preserving upstream's `pre_mix` condition for Gram computation. | +| Embedded GPU policies | Decompressed three-way comparison passes for WS, Max-Q and GB10. Each retains the R34 MHC component and the upstream/unchanged values of every other component; no unresolved policy field. | +| B12X tests/evidence | Uses upstream's split-kernel tests and benchmark evidence. Four Qwen/recipe guard cases pass when replayed from R34. Historical positional fake-operator calls require the installed schemas; the recipe provides 15 keyword/schema-based CPU checks. | +| LMCache | Zero residual differences; complete source tree equals R34. | + +The positional fake-dispatch tests are not evidence of a CUDA failure: two +historical calls reject the composed operator's extended argument schema +before execution. The corresponding schema-bound tests pass without changing +the operators or weakening an arithmetic tolerance. + +## Runtime and packaging preservation + +- Both images have **two filesystem layers** and the exact same runtime-base + DiffID, `sha256:cb220f94ebc323749ae8e960eecd0076d01b4a23ee7d2dabdb463b44fdc37d68`. +- Launch scripts, entrypoint, command, ports, healthcheck and volume settings + are identical. Eleven environment differences identify source-derived JIT + cache namespaces; serving defaults do not change. +- Native library overrides are byte-identical. The shared base and unchanged + overrides preserve CUDA, PyTorch, FlashInfer, FlashKDA and compatible native + extensions. No native library was substituted by this composition. +- Outside vLLM/B12X, differences are build-cache files, generated bytecode, + source-lock metadata and LMCache wheel/build timestamps. The installed + LMCache native and Python payloads are identical despite a different wheel + archive hash. The wheel `RECORD` changes only its `uv_cache.json` timestamp + entry. +- Of the differing bytecode entries, 791 differ only in their header while + their source is identical; 34 accompany changed/added source. There are no + unexplained bytecode differences. Generated vLLM version metadata identifies + the respective source revision. + +## SwiGLU correctness proof + +Derek Yates (D-Rock) reported that R33's split NVFP4 phase-1 kernel omitted +the model's SwiGLU limit. R34 uses the same B12X revision. GLM's configured +limit is 10: clamp the gate above the limit and the up projection on both +sides before `silu(gate) * up`. Omitting this changes model semantics even +when all outputs remain finite. + +An independent test on RTX PRO 6000 Workstation GPU +`GPU-a2f726b2-60db-8102-abef-abf76c457fe0`, SM120, driver 610.57.04, exercises +production fused-MoE dispatch. It uses a tight limit of 2 to guarantee clamping +on the synthetic domain (512 rows, 8 experts, K256, N128, top-k 2, seed 42). +Unclamped and clamped outputs are checked against their respective NVFP4 +oracles; the gates require finite output, cosine above 0.9999, BF16 error +bounds and a measurable clamp effect. + +| Arm | Result | +|---|---| +| Immutable R34, unmodified runtime | **FAIL:** cosine 0.929721, RMSE 0.365815, maximum absolute error 3.113281 | +| Same R34 with only 13 clamp/parameter-propagation lines in two B12X files | **PASS** | +| Immutable composed image, unmodified runtime | **PASS** | + +Raw test receipts: [R34 failure](evidence/r34-preservation/r34-swiglu-gpu.log), +[13-line-only correction](evidence/r34-preservation/r34-minimal-swiglu-gpu.log), +[composed-image clamp](evidence/r34-preservation/composition-swiglu-gpu.log), +[four GPU dispatch cases](evidence/r34-preservation/composition-dispatch-gpu.log), +and [15 CPU contracts](evidence/r34-preservation/composition-contracts-cpu.log). + +The corrected B12X merge is `8648fae3bc19c164b46098c1e97882efd8443876`, an +ancestor of the composed revision. The split-materialized path stays enabled; +the fix does not switch to a slower backend or add another kernel launch. + +The composed image also passes all four production split-dispatch tests, +including CUDA-graph/scratch reuse and unsupported-domain behavior. Fifteen +CPU graph-interface/quantization-guard checks pass. No model server or GPU +clock configuration was changed for these focused checks. + +### Mandatory image gate + +Inside the B12X source directory of an image with an assigned test GPU, run: + +```bash +/opt/venv/bin/python -m pytest -p no:cacheprovider -q \ + tests/moe/test_nvfp4_split_dispatch_policy.py::TestNvfp4SplitDispatch +``` + +Mount [the recipe's CPU contract check](../../tests/verify_b12x_release_contracts.py) +read-only and run it with the same image's Python. No model download is needed. +Do not accept a build solely because `swiglu_limit` appears in the source or +because a finite-output smoke test passes. + +## Qualification limits + +The [matched R34/composition serving measurements](review-qualification.md) +remain the performance evidence: DFlash2 K7 TP4/DCP1 C1, C8 and cold 32K +prefill meet the stated bounded 2% screen. They are not a complete performance +matrix, and R34 is not a valid clamped-math correctness oracle. + +The GPU reproducer proves the clamp defect and its correction. It does **not** +prove that every reported 20K–168K runaway has that cause. D-Rock's production- +shaped churn evidence is separately reported evidence, not a soak repeated by +this audit. A field replay or controlled production canary is still needed to +establish causality for those incidents. No claim is made that LMCache itself +caused the incorrect activations. + +The published R34 digest remains unchanged and contains the defect. The source +composition is a corrected build input, not a retroactive correction of R34. diff --git a/recipes/glm53/reasoning-high.source.lock b/recipes/glm53/reasoning-high.source.lock new file mode 100644 index 00000000..0dadf33b --- /dev/null +++ b/recipes/glm53/reasoning-high.source.lock @@ -0,0 +1,52 @@ +format=local-inference-source-lock/v1 +implementation.status=implemented +qualification.contract=See evidence for the resulting immutable image +release.name=jovian-judgement-community-20260908-r28-high +release.version=r28-high +runtime.base.image=voipmonitor/vllm@sha256:93ac5228f1cbde2182ca294d8b479259144742af2756a49ff207dd245429bf43 +runtime.cuda.version=13.3 +runtime.pytorch.version=2.13.0 +runtime.rootfs.layers=3 +runtime.cudagraph.mode=FULL_AND_PIECEWISE +runtime.scheduler.max-num-batched-tokens=4096 +runtime.scheduler.prefill-compute-share=0.4 +runtime.nccl.channels=16 +runtime.nccl.buffer.bytes=2097152 +runtime.recurrent-checkpoint-policy=auto +runtime.lmcache.transfer=engine-driven asynchronous shared memory +runtime.lmcache.checkpoints=atomic target and draft bundles for request_boundaries; independent chunks for explicit aligned +model.repository=local-inference-lab/GLM-5.3-Flash-NVFP4 +draft.repository=local-inference-lab/GLM-5.3-Flash-DFlash2 +draft.quantization=MXFP8 +vllm.commit=2531689fa50b956d3e1156e1ab80d119aaf34c1e +vllm.tree=a8b355eb38a0b75b97eae06738b03c65b1e9fbb5 +vllm.package.tree=b58e8822b7f01df715fe72a94ed9ae4d2f447040 +vllm.bundle.sha256=76d4654ce69e6dedd2e0b4bd83155dd6306e19ef5254bb67d29a96acb67372c3 +b12x.commit=3edbcbce70f491741b82f5eab9c1b30b39447228 +b12x.tree=c4bfeee9f3c9457400d191c870eb2e44fbcd5c2e +b12x.package.tree=d76de6c4ceaa9c2feb49edaf8c2f77d5a29ab5cb +b12x.bundle.sha256=a3ef23b6d179ec95b8a160cf509d4c75d6b90c58b5a604c47ce8b58ecf786e44 +lmcache.commit=617a1b47a790de6b86eea92f59deb232a9eff87d +lmcache.tree=247f81b6fd3b156c402a0dd31cd4a21ae8e21160 +lmcache.package.tree=85622f21a2a3ae115e7a4ca41801fe221001e87f +lmcache.bundle.sha256=1ab4ac35c21e7fb623799acb49ce8fea25a39ea1b18751b5624c26b025f74e55 +lmcache.version=0.5.5.dev0+glm53checkpoints.617a1b47 +flashkda.base.commit=3b225bf26bb8e218928a1fe14751cb48cf31d11b +flashkda.patch.sha256=a9537532cf231babc942fa592666893268ca0617b24daeb8e78d3249e8c1d58a +flashkda.extension.sha256=3bb6ef2c9f0be24b2c6ae8a48eb20b53f28e632ed9cb45a32c7c0133f8cfde91 +input.Dockerfile.glm53-cache-contracts.sha256=0431cf8cf4f9979423e60cea8ce53a940347a7f3c2d70440d6c5bf123fc98f76 +input.build_glm53_cache_contract_image.sh.sha256=9ee13a42b0ad78b9f0c5e97dd0a376960ed4e80c484b708a9e54db5570da75ae +input.source_locked_image_labels.py.sha256=b28378354056ab145edc8761dd232e3e2aab2bb691fdaa6b5175481d941e894a +input.install_source_bundle.sh.sha256=d18c4b49acc40722b439a0838240c5a03c5df5b799027cd3c9c141fb78395613 +input.install_glm53_source_locked.sh.sha256=429cee897017343f129f65200aa829f1cf7cfd6c6f04897c724a467873849b1b +input.serve-glm53-flash-nvfp4-dflash2.sh.sha256=856f1a25fc2524a543983b32a291efa8e230705ca79539bd7e5a139dd39489e2 +input.serve-glm53-flash-nvfp4-dflash2-scheduler-qos.sh.sha256=69b0bf28564d5f497d6da1d6eb34ba45eb505be788247e8acce793bd88497d69 +input.serve-glm53-flash-lmcache.sh.sha256=c841af83547ecd6fd4bcf04cd8933b9f894eb5c9e27421130938b1168c0ac0c1 +input.serve-glm53-flash-lmcache-cache-complete.sh.sha256=2e2e652c5f56ddfad6b123b24d938d6bcb62fcb2c1ea90a8b990ac3714b511f7 +input.serve-glm53-flash-cache-complete.sh.sha256=167da2e8e8ca79d8c50d93d279d2526b133737cfad8322ecca20c684137749f0 +input.glm53_checkpoint_identity.py.sha256=b5d259c38bf844d4da6a1a9d905a0e8983f586a8dc550b138d8cdf070c02f499 +input.glm53-r18-lmcache-runtime-requirements.txt.sha256=7ac402540e2ba3f02fc1a443934ef3ff6e2e10c8ff81ae989e3958cfff38adf7 +build.uv.sha256=9a4299a0c3bcc01012acbcdae7b5655e5087e0e5d87459306736a1420a902b81 +runtime.cache.fingerprint=cu133-torch213-glm53-vllmb58e8822-b12xd76de6c4-lmcache85622f21-flashkda3bb6ef2c +runtime.default.reasoning-effort=high +runtime.parent.release-image=voipmonitor/vllm@sha256:f5f121e37fd2afbb6f8f036e7eb627435cfb736de0a4420306dc2a25b6631669 diff --git a/recipes/glm53/review-composition.md b/recipes/glm53/review-composition.md new file mode 100644 index 00000000..dc4d7cd9 --- /dev/null +++ b/recipes/glm53/review-composition.md @@ -0,0 +1,96 @@ +# Public review composition for Jovian serving + +Status: **implemented; exact source-tree composition qualified**. The bounded +[GPU serving evidence and source mirrors](review-qualification.md) are recorded +separately; serving correctness does not follow merely from a successful merge. + +The manifests in this directory describe complete vLLM, B12X and LMCache source +trees as a pinned base plus ordered public pull-request heads. They contain no +`source_patches` or private conflict-resolution input. A review head owns every +required resolution. The resulting tree is checked against the manifest's +`expected_tree` before an integration artifact is accepted. + +| Component | Base branch | Base commit | Manifest | +|---|---|---|---| +| vLLM | `dev/jovian-judgement` | `a6571d0a602ed42525a7a77cf960f39ca7f8d7a8` | [32 review units](review-stack.vllm.json) | +| B12X | `master` | `00b69ac22e21413622c4ecd98f607a2c3e015161` | [3 review units](review-stack.b12x.json) | +| LMCache | `dev` | `7ed4675404a31f4ffafd98975899dc83832ba965` | [9 review units](review-stack.lmcache.json) | + +## Verification + +Run from the repository root: + +```bash +uv run --no-project scripts/compose_vllm_release.py \ + recipes/glm53/review-stack.vllm.json --output-dir ./review-vllm +uv run --no-project scripts/compose_vllm_release.py \ + recipes/glm53/review-stack.b12x.json --output-dir ./review-b12x +uv run --no-project scripts/compose_vllm_release.py \ + recipes/glm53/review-stack.lmcache.json --output-dir ./review-lmcache +``` + +The composer fetches public GitHub PR refs, rejects moved heads or unresolved +dependencies, merges in the declared order and verifies the complete Git tree. +It emits a generated `integration.patch` and `integration.lock.json`; the patch +is an output, not an extra integration input. Patch replay is checked against +the same tree. Embedded `.patch` files retain meaningful unified-diff context +spaces; ordinary source files remain subject to whitespace validation. + +For a checkout whose object database already contains the pinned commits, the +offline check does not edit its branch, index or working files: + +```bash +uv run --no-project scripts/compose_vllm_release.py \ + recipes/glm53/review-stack.vllm.json --verify-in ./vllm-source +``` + +An offline proof checks composition only, not publication. Use the network +verification when accepting public review heads. The declared order is a +supported composition contract, not a claim that all subsets or merge orders +are valid. + +## Review ownership + +- vLLM #664 owns scheduling, checkpoint-readiness polling and prefill-lane + release. Apply #664 and #553 before #709's atomic checkpoint import. +- vLLM #729 preserves logprobz's local DCP1 checkpoint admission implementation + and supersedes #721. Apply #718 first. The local admission feature remains + disabled for DCP4 and external connectors. +- vLLM #708 retains packed recurrent export and the checkpoint-restore test + above a 2 GiB byte offset; #676 remains the separate partial-prefix proof. +- B12X #356 retains MadeBy561's capacity-based MHC policy. It is not the + source-split scheduling proposal in #310. The policy preserves the master + schemas and non-MHC profile components. +- LMCache #49 incorporates Derek Yates's filesystem ledger contribution from + #67 together with bounded object names. Do not apply #67 separately. +- B12X #353/#354 and the behavior reviewed in #317 are represented in the + pinned master source. vLLM #727 is represented in pinned JJ. They are not + additional merge inputs; inclusion alone does not imply that GitHub marked + each original PR merged. + +Merge commits preserve contributor history. Source bundles must be built from +the complete attributed Git composition, not a squashed filesystem copy. The +two-layer Docker recipe authenticates the resulting source trees and compatible +native binaries independently of the PR titles. + +## Compatibility boundary + +The serving composition retains the R34 cache, speculative decoding, model +launchers, sampling defaults and B12X deployment default while incorporating +the declared JJ/master additions. LMCache's composed tree is exactly the R34 +tree. No vLLM native ABI or CUDA runtime change is required by these JJ merges. + +vLLM #730 isolates quantization configuration discovery from model-specific +native layer imports. JJ's standalone B12X imports are preserved; actual +DeepSeek V4.1 dispatch still requires its native backend. Import isolation is +not a fallback implementation or a DeepSeek V4.1 serving qualification. + +The published R34 image and the source identities in the preservation audit are +immutable reference artifacts. These review manifests and the [build guide](README.md) +identify the corrected composition. They do not change the contents of an +already-published image. + +The [installed-artifact and R34 preservation audit](r34-preservation.md) verifies +this compatibility statement independently of composition replay. It also +proves that the pinned B12X base fixes R34's omitted split-NVFP4 SwiGLU clamp. +R34 is a performance reference, not a valid oracle for that clamped operation. diff --git a/recipes/glm53/review-qualification.json b/recipes/glm53/review-qualification.json new file mode 100644 index 00000000..0541f3a8 --- /dev/null +++ b/recipes/glm53/review-qualification.json @@ -0,0 +1,954 @@ +{ + "r34": { + "decode": { + "metadata": { + "version": "0.4.29", + "engine": "vllm", + "model": "GLM-5.3-Flash-NVFP4", + "server": "127.0.0.1:5051", + "timestamp": "2026-09-11T00:23:26.823786", + "decode_mode": "duration", + "primary_decode_layer": "sustained_decode", + "duration_per_test": 30, + "request_count": 0, + "warmup_request_count": 0, + "run_burst": false, + "prefill_mode": "skipped", + "standalone_prefill": false, + "prefill_only": false, + "skip_prefill": true, + "burst_e2e_status": "not_run_use_--run-burst", + "burst_request_count": 0, + "burst_warmup_request_count": 0, + "burst_requests_per_concurrency": 5, + "decode_warmup_seconds": 15, + "decode_warmup_context": 0, + "decode_warmup_concurrency": 1, + "cell_warmup_timeout_seconds": 0, + "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0", + "show_capacity_limited_values": false, + "max_tokens": 8192, + "temperature": 1, + "top_p": 0.95, + "ignore_eos": true, + "max_total_tokens": 3977216, + "dcp_size": 0, + "metrics_available": true, + "metrics_warning": "", + "concurrency_levels": [ + 1, + 8 + ], + "context_lengths": [ + 0 + ], + "startup_diagnostics_available": true, + "nvidia_p2p_override_effective": true, + "p2pmark_status": "not_run", + "amd_fabric_status": "not_run" + }, + "results": [ + { + "concurrency": 1, + "context_length": null, + "aggregate_tps": 254.85236863112578, + "server_steps_per_s": 97.37393966926335, + "server_accept_len_effective": 2.6172543649435127, + "num_errors": 0, + "total_tokens": 7645, + "total_duration": null + }, + { + "concurrency": 8, + "context_length": null, + "aggregate_tps": 803.7028635485832, + "server_steps_per_s": 308.7370761284655, + "server_accept_len_effective": 2.603195164075993, + "num_errors": 0, + "total_tokens": 24092, + "total_duration": null + } + ] + }, + "prefill": { + "conditions": { + "prompt_tokens": 32768, + "max_tokens": 1, + "duration_seconds": 30, + "temperature": 1, + "top_p": 0.95, + "prompt_type": "deterministic cyclic token IDs with unique leading nonce" + }, + "warmup": { + "nonce": 1789086206961314600, + "prompt_tokens": 32768, + "ttft_seconds": 1.935370072023943, + "tok_per_sec": 16931.128818031353, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + "samples": [ + { + "nonce": 1789086208915468500, + "prompt_tokens": 32768, + "ttft_seconds": 1.9395071050385013, + "tok_per_sec": 16895.014158429454, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086210863064000, + "prompt_tokens": 32768, + "ttft_seconds": 1.932527094031684, + "tok_per_sec": 16956.0365291638, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086212804968700, + "prompt_tokens": 32768, + "ttft_seconds": 1.940648616058752, + "tok_per_sec": 16885.07632388818, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086214753635800, + "prompt_tokens": 32768, + "ttft_seconds": 1.9413468079874292, + "tok_per_sec": 16879.003723178233, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086216703075600, + "prompt_tokens": 32768, + "ttft_seconds": 1.9399111530510709, + "tok_per_sec": 16891.49523598689, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086218651018800, + "prompt_tokens": 32768, + "ttft_seconds": 1.939960754942149, + "tok_per_sec": 16891.063345751634, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086220598984200, + "prompt_tokens": 32768, + "ttft_seconds": 1.9396944979671389, + "tok_per_sec": 16893.381939445568, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086222546756900, + "prompt_tokens": 32768, + "ttft_seconds": 1.9381978450110182, + "tok_per_sec": 16906.42680485166, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086224492911400, + "prompt_tokens": 32768, + "ttft_seconds": 1.9439705229597166, + "tok_per_sec": 16856.222670552823, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086226444828200, + "prompt_tokens": 32768, + "ttft_seconds": 1.9458797010593116, + "tok_per_sec": 16839.684376254878, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086228398666000, + "prompt_tokens": 32768, + "ttft_seconds": 1.9437761880690232, + "tok_per_sec": 16857.90792228617, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086230350383400, + "prompt_tokens": 32768, + "ttft_seconds": 1.9449518079636618, + "tok_per_sec": 16847.718213803793, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086232303267300, + "prompt_tokens": 32768, + "ttft_seconds": 1.995293966960162, + "tok_per_sec": 16422.642749690753, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086234308744700, + "prompt_tokens": 32768, + "ttft_seconds": 1.9605366340838373, + "tok_per_sec": 16713.791229569426, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086236277418800, + "prompt_tokens": 32768, + "ttft_seconds": 1.9429342689691111, + "tok_per_sec": 16865.21285014246, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789086238228521200, + "prompt_tokens": 32768, + "ttft_seconds": 1.9475057600066066, + "tok_per_sec": 16825.624176787464, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + } + ], + "median_tok_per_sec": 16872.108286660346, + "status": "qualified" + } + }, + "reviewed": { + "decode": { + "metadata": { + "version": "0.4.29", + "engine": "vllm", + "model": "GLM-5.3-Flash-NVFP4", + "server": "127.0.0.1:5051", + "timestamp": "2026-09-11T00:51:06.596320", + "decode_mode": "duration", + "primary_decode_layer": "sustained_decode", + "duration_per_test": 30, + "request_count": 0, + "warmup_request_count": 0, + "run_burst": false, + "prefill_mode": "skipped", + "standalone_prefill": false, + "prefill_only": false, + "skip_prefill": true, + "burst_e2e_status": "not_run_use_--run-burst", + "burst_request_count": 0, + "burst_warmup_request_count": 0, + "burst_requests_per_concurrency": 5, + "decode_warmup_seconds": 15, + "decode_warmup_context": 0, + "decode_warmup_concurrency": 1, + "cell_warmup_timeout_seconds": 0, + "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0", + "show_capacity_limited_values": false, + "max_tokens": 8192, + "temperature": 1, + "top_p": 0.95, + "ignore_eos": true, + "max_total_tokens": 3985408, + "dcp_size": 0, + "metrics_available": true, + "metrics_warning": "", + "concurrency_levels": [ + 1, + 8 + ], + "context_lengths": [ + 0 + ], + "startup_diagnostics_available": true, + "nvidia_p2p_override_effective": true, + "p2pmark_status": "not_run", + "amd_fabric_status": "not_run" + }, + "results": [ + { + "concurrency": 1, + "context_length": null, + "aggregate_tps": 255.0536080632398, + "server_steps_per_s": 97.70776698906755, + "server_accept_len_effective": 2.610371886728079, + "num_errors": 0, + "total_tokens": 7651, + "total_duration": null + }, + { + "concurrency": 8, + "context_length": null, + "aggregate_tps": 792.8434317173468, + "server_steps_per_s": 310.7359454838382, + "server_accept_len_effective": 2.551502145922747, + "num_errors": 0, + "total_tokens": 23780, + "total_duration": null + } + ] + }, + "prefill": { + "conditions": { + "prompt_tokens": 32768, + "max_tokens": 1, + "duration_seconds": 30, + "temperature": 1, + "top_p": 0.95, + "prompt_type": "deterministic cyclic token IDs with unique leading nonce" + }, + "warmup": { + "nonce": 1789087883252008700, + "prompt_tokens": 32768, + "ttft_seconds": 4.114826904027723, + "tok_per_sec": 7963.3969457927005, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + "samples": [ + { + "nonce": 1789087887385593300, + "prompt_tokens": 32768, + "ttft_seconds": 1.954206472961232, + "tok_per_sec": 16767.931359037135, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087889349519400, + "prompt_tokens": 32768, + "ttft_seconds": 1.960569340037182, + "tok_per_sec": 16713.512412358014, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087891319399000, + "prompt_tokens": 32768, + "ttft_seconds": 1.9628037629881874, + "tok_per_sec": 16694.4860295732, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087893290444000, + "prompt_tokens": 32768, + "ttft_seconds": 1.963144172099419, + "tok_per_sec": 16691.59120644581, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087895262220000, + "prompt_tokens": 32768, + "ttft_seconds": 1.960344123071991, + "tok_per_sec": 16715.432568364755, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087897230648300, + "prompt_tokens": 32768, + "ttft_seconds": 1.963066834025085, + "tok_per_sec": 16692.24879766945, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087899202194200, + "prompt_tokens": 32768, + "ttft_seconds": 1.962875035009347, + "tok_per_sec": 16693.879852541893, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087901173437200, + "prompt_tokens": 32768, + "ttft_seconds": 1.9628464760025963, + "tok_per_sec": 16694.122745011187, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087903144436500, + "prompt_tokens": 32768, + "ttft_seconds": 1.9642949110129848, + "tok_per_sec": 16681.81280533969, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087905117042700, + "prompt_tokens": 32768, + "ttft_seconds": 1.9618635710794479, + "tok_per_sec": 16702.48659644082, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087907087240400, + "prompt_tokens": 32768, + "ttft_seconds": 1.9648613509489223, + "tok_per_sec": 16677.003689942205, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087909060780300, + "prompt_tokens": 32768, + "ttft_seconds": 1.9603312369436026, + "tok_per_sec": 16715.542446331336, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087911030137000, + "prompt_tokens": 32768, + "ttft_seconds": 1.9611774049699306, + "tok_per_sec": 16708.330371827025, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087912999755000, + "prompt_tokens": 32768, + "ttft_seconds": 1.9666655550245196, + "tok_per_sec": 16661.70433314548, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087914974735600, + "prompt_tokens": 32768, + "ttft_seconds": 1.9679556098999456, + "tok_per_sec": 16650.782078192293, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + }, + { + "nonce": 1789087916951012600, + "prompt_tokens": 32768, + "ttft_seconds": 1.9589756439672783, + "tok_per_sec": 16727.10944666923, + "prompt_sources": { + "external_kv_transfer": 0, + "local_compute": 32768, + "local_cache_hit": 0 + } + } + ], + "median_tok_per_sec": 16694.30438729219, + "status": "qualified" + } + }, + "prefix": { + "passed": true, + "cases": [ + { + "name": "exact-8192-run-0", + "elapsed_seconds": 0.6471811649389565, + "usage": { + "prompt_tokens": 8192, + "total_tokens": 8320, + "completion_tokens": 128, + "prompt_tokens_details": null, + "completion_tokens_details": null + }, + "metrics": { + "prefix_hits": 0, + "prefix_queries": 8192, + "drafts": 16, + "draft_tokens": 112, + "accepted_tokens": 111, + "preemptions": 0 + }, + "accepted_per_draft_step": 6.9375, + "matches_first_output": true + }, + { + "name": "exact-8192-run-1", + "elapsed_seconds": 0.16948784398846328, + "usage": { + "prompt_tokens": 8192, + "total_tokens": 8320, + "completion_tokens": 128, + "prompt_tokens_details": null, + "completion_tokens_details": null + }, + "metrics": { + "prefix_hits": 8192, + "prefix_queries": 8192, + "drafts": 16, + "draft_tokens": 112, + "accepted_tokens": 111, + "preemptions": 0 + }, + "accepted_per_draft_step": 6.9375, + "matches_first_output": true, + "restored_exact_endpoint": true + }, + { + "name": "exact-8192-run-2", + "elapsed_seconds": 0.16902486502658576, + "usage": { + "prompt_tokens": 8192, + "total_tokens": 8320, + "completion_tokens": 128, + "prompt_tokens_details": null, + "completion_tokens_details": null + }, + "metrics": { + "prefix_hits": 8192, + "prefix_queries": 8192, + "drafts": 16, + "draft_tokens": 112, + "accepted_tokens": 111, + "preemptions": 0 + }, + "accepted_per_draft_step": 6.9375, + "matches_first_output": true, + "restored_exact_endpoint": true + }, + { + "name": "response-continuation-8192", + "elapsed_seconds": 0.36602806102018803, + "usage": { + "prompt_tokens": 8325, + "total_tokens": 8453, + "completion_tokens": 128, + "prompt_tokens_details": null, + "completion_tokens_details": null + }, + "metrics": { + "prefix_hits": 8319, + "prefix_queries": 8325, + "drafts": 16, + "draft_tokens": 112, + "accepted_tokens": 111, + "preemptions": 0 + }, + "accepted_per_draft_step": 6.9375, + "matches_cold_output": true, + "restored_response_endpoint": true + }, + { + "name": "response-continuation-8192-cold-control", + "elapsed_seconds": 0.6747887689853087, + "usage": { + "prompt_tokens": 8325, + "total_tokens": 8453, + "completion_tokens": 128, + "prompt_tokens_details": null, + "completion_tokens_details": null + }, + "metrics": { + "prefix_hits": 0, + "prefix_queries": 8325, + "drafts": 16, + "draft_tokens": 112, + "accepted_tokens": 111, + "preemptions": 0 + }, + "accepted_per_draft_step": 6.9375 + }, + { + "name": "instruction-cold", + "elapsed_seconds": 1.0398785159923136, + "usage": { + "prompt_tokens": 11351, + "total_tokens": 11367, + "completion_tokens": 16, + "prompt_tokens_details": null, + "completion_tokens_details": { + "reasoning_tokens": 9 + } + }, + "metrics": { + "prefix_hits": 0, + "prefix_queries": 11351, + "drafts": 4, + "draft_tokens": 28, + "accepted_tokens": 11, + "preemptions": 0 + }, + "accepted_per_draft_step": 2.75, + "expected": "VALUE_2342", + "correct": true + }, + { + "name": "instruction-sibling", + "elapsed_seconds": 0.08777603600174189, + "usage": { + "prompt_tokens": 11351, + "total_tokens": 11358, + "completion_tokens": 7, + "prompt_tokens_details": null, + "completion_tokens_details": { + "reasoning_tokens": 0 + } + }, + "metrics": { + "prefix_hits": 11340, + "prefix_queries": 11351, + "drafts": 2, + "draft_tokens": 14, + "accepted_tokens": 4, + "preemptions": 0 + }, + "accepted_per_draft_step": 2, + "expected": "VALUE_3511", + "correct": true + }, + { + "name": "instruction-sibling-replay", + "elapsed_seconds": 0.05524280306417495, + "usage": { + "prompt_tokens": 11351, + "total_tokens": 11358, + "completion_tokens": 7, + "prompt_tokens_details": null, + "completion_tokens_details": { + "reasoning_tokens": 0 + } + }, + "metrics": { + "prefix_hits": 11351, + "prefix_queries": 11351, + "drafts": 2, + "draft_tokens": 14, + "accepted_tokens": 4, + "preemptions": 0 + }, + "accepted_per_draft_step": 2, + "expected": "VALUE_3511", + "correct": true + }, + { + "name": "instruction-sibling-cold-control", + "elapsed_seconds": 0.7640913829673082, + "usage": { + "prompt_tokens": 11351, + "total_tokens": 11358, + "completion_tokens": 7, + "prompt_tokens_details": null, + "completion_tokens_details": { + "reasoning_tokens": 0 + } + }, + "metrics": { + "prefix_hits": 0, + "prefix_queries": 11351, + "drafts": 2, + "draft_tokens": 14, + "accepted_tokens": 4, + "preemptions": 0 + }, + "accepted_per_draft_step": 2, + "expected": "VALUE_3511", + "correct": true + }, + { + "name": "concurrent-instructions-run-0", + "clients": 4, + "elapsed_seconds": 1.921979657956399, + "correct": true, + "metrics": { + "prefix_hits": 22680, + "prefix_queries": 45404, + "drafts": 8, + "draft_tokens": 56, + "accepted_tokens": 19, + "preemptions": 0 + } + }, + { + "name": "concurrent-instructions-run-1", + "clients": 4, + "elapsed_seconds": 0.14362517895642668, + "correct": true, + "metrics": { + "prefix_hits": 45404, + "prefix_queries": 45404, + "drafts": 8, + "draft_tokens": 56, + "accepted_tokens": 19, + "preemptions": 0 + } + }, + { + "name": "chat-first-turn", + "elapsed_seconds": 0.7751922989264131, + "usage": { + "prompt_tokens": 11351, + "total_tokens": 11367, + "completion_tokens": 16, + "prompt_tokens_details": null, + "completion_tokens_details": { + "reasoning_tokens": 9 + } + }, + "metrics": { + "prefix_hits": 0, + "prefix_queries": 11351, + "drafts": 3, + "draft_tokens": 21, + "accepted_tokens": 12, + "preemptions": 0 + }, + "accepted_per_draft_step": 4 + }, + { + "name": "chat-assistant-continuation", + "elapsed_seconds": 0.13198545994237065, + "usage": { + "prompt_tokens": 11378, + "total_tokens": 11394, + "completion_tokens": 16, + "prompt_tokens_details": null, + "completion_tokens_details": { + "reasoning_tokens": 9 + } + }, + "metrics": { + "prefix_hits": 11366, + "prefix_queries": 11378, + "drafts": 3, + "draft_tokens": 21, + "accepted_tokens": 12, + "preemptions": 0 + }, + "accepted_per_draft_step": 4, + "expected": "VALUE_3511", + "correct": true + }, + { + "name": "chat-assistant-continuation-cold-control", + "elapsed_seconds": 0.7935983749339357, + "usage": { + "prompt_tokens": 11378, + "total_tokens": 11394, + "completion_tokens": 16, + "prompt_tokens_details": null, + "completion_tokens_details": { + "reasoning_tokens": 9 + } + }, + "metrics": { + "prefix_hits": 0, + "prefix_queries": 11378, + "drafts": 3, + "draft_tokens": 21, + "accepted_tokens": 12, + "preemptions": 0 + }, + "accepted_per_draft_step": 4, + "expected": "VALUE_3511", + "correct": true + }, + { + "name": "chat-tool-history-run-0", + "elapsed_seconds": 0.7833004230633378, + "usage": { + "prompt_tokens": 11520, + "total_tokens": 11527, + "completion_tokens": 7, + "prompt_tokens_details": null, + "completion_tokens_details": { + "reasoning_tokens": 0 + } + }, + "metrics": { + "prefix_hits": 0, + "prefix_queries": 11520, + "drafts": 1, + "draft_tokens": 7, + "accepted_tokens": 5, + "preemptions": 0 + }, + "accepted_per_draft_step": 5, + "expected": "VALUE_3511", + "correct": true + }, + { + "name": "chat-tool-history-run-1", + "elapsed_seconds": 0.04661498498171568, + "usage": { + "prompt_tokens": 11520, + "total_tokens": 11527, + "completion_tokens": 7, + "prompt_tokens_details": null, + "completion_tokens_details": { + "reasoning_tokens": 0 + } + }, + "metrics": { + "prefix_hits": 11520, + "prefix_queries": 11520, + "drafts": 1, + "draft_tokens": 7, + "accepted_tokens": 5, + "preemptions": 0 + }, + "accepted_per_draft_step": 5, + "expected": "VALUE_3511", + "correct": true + }, + { + "name": "chat-divergent-instructions", + "elapsed_seconds": 0.7645809520035982, + "usage": { + "prompt_tokens": 11348, + "total_tokens": 11352, + "completion_tokens": 4, + "prompt_tokens_details": null, + "completion_tokens_details": { + "reasoning_tokens": 0 + } + }, + "metrics": { + "prefix_hits": 0, + "prefix_queries": 11348, + "drafts": 2, + "draft_tokens": 14, + "accepted_tokens": 1, + "preemptions": 0 + }, + "accepted_per_draft_step": 0.5, + "expected": "VALUE_CHANGED", + "correct": true + } + ] + }, + "contract": { + "status": "qualified", + "qualification_scope": "TP4/DCP1 DFlash2 K7 FP8-target-KV serving and prefix smoke; not a full release matrix", + "vllm_commit": "de982a50c6a3e4718e5cf9f00423a92192718da1", + "b12x_commit": "98086604c86ec1e78977e5023ce282ecb97ab8a7", + "lmcache_commit": "29bc5a2efde737c436b04499eb62cd1776cebeec", + "tested_image_id": "sha256:d3508d5afd3c616db55efb5d18490e12370fdd0f9c6821f434136e4f4fc21c6a", + "reference_image_digest": "sha256:d2d13141fc158f3e5f989930c4be4637eaf28322307288724c2faa7cd4e9bcc7", + "source_lock_sha256": "7be25b839e3c34b68e5f54dbec6d12b2cb540277966e49c424e082e57c448e88", + "gpu_indices": [ + 4, + 5, + 6, + 7 + ], + "vram_offset_mhz": 6000, + "graphics_offset_mhz": 0, + "power_limit_watts": 600, + "attention_backend": "B12X", + "moe_backend": "b12x", + "kda_prefill_backend": "flashkda", + "model_runner": "v2", + "graph_mode": "FULL_AND_PIECEWISE", + "scheduler_tokens": 4096, + "omp_threads": 1, + "nccl_channels": 16, + "nccl_buffer_bytes": 2097152 + } +} diff --git a/recipes/glm53/review-qualification.md b/recipes/glm53/review-qualification.md new file mode 100644 index 00000000..7113dc25 --- /dev/null +++ b/recipes/glm53/review-qualification.md @@ -0,0 +1,86 @@ +# Qualification of the public Jovian source composition + +Status: **qualified for the bounded checks below**, measured 2026-09-11. +This is not a DockerHub release or a full GLM/Qwen/DeepSeek qualification. + +The separate [R34 preservation and activation-correctness audit](r34-preservation.md) +checks actual installed files, retained runtime behavior and the split-NVFP4 +SwiGLU correction. R34 fails that clamp regression; the composed image passes. + +The [review manifests](review-composition.md) reproduce complete source trees +from public GitHub refs with no additional patches. Network fetch, ordered +merge and generated-patch replay passed for 32 vLLM, 3 B12X and 9 LMCache +review units. All 44 were open and non-draft at the audit. + +## Attributed source mirrors + +| Component and review repository | Complete source mirror | Tested commit | +|---|---|---| +| vLLM — [local-inference-lab/vllm](https://github.com/local-inference-lab/vllm) | [voipmonitor/vllm](https://github.com/voipmonitor/vllm/tree/integration/jovian-reviewed-sources-20260911) | `de982a50c6a3e4718e5cf9f00423a92192718da1` | +| B12X — [local-inference-lab/b12x](https://github.com/local-inference-lab/b12x) | [voipmonitor/b12x](https://github.com/voipmonitor/b12x/tree/integration/jovian-reviewed-sources-20260911) | `98086604c86ec1e78977e5023ce282ecb97ab8a7` | +| LMCache — [local-inference-lab/LMCache](https://github.com/local-inference-lab/LMCache) | [local-inference-lab/LMCache](https://github.com/local-inference-lab/LMCache/tree/release/jovian-fp4-fs-ledger-r33-20260910) | `29bc5a2efde737c436b04499eb62cd1776cebeec` | + +The candidate has two filesystem layers and source-lock SHA-256 +`7be25b839e3c34b68e5f54dbec6d12b2cb540277966e49c424e082e57c448e88`. +Its local image ID and the public R34 digest are recorded with +[measured cells, prefill samples and prefix checks](review-qualification.json). +CUDA, FlashInfer and compatible native artifacts are unchanged. Serving used +no source bind mounts. + +## Matched serving comparison + +Conditions: the same physical RTX PRO 6000 Workstation GPUs 4–7, TP4/DCP1, +VRAM offset **+6000 MHz**, graphics offset 0, 600 W limit. These are not stock +measurements. Both images use the same target and MXFP8 draft snapshots, +DFlash2 K7 probabilistic/standard sampling, temperature 1, top-p 0.95, +FP8 target KV, B12X MoE/attention/all-reduce, FlashKDA recurrent prefill, +V2 runner, 4096-token scheduler budget, OMP1, NCCL16/2MiB and +`FULL_AND_PIECEWISE` graphs. + +| Metric | Published R34 | Public review composition | Change | +|---|---:|---:|---:| +| C1 output tok/s | 254.85 | 255.05 | +0.08% | +| C1 verifier steps/s | 97.37 | 97.71 | +0.34% | +| C8 aggregate output tok/s | 803.70 | 792.84 | −1.35% | +| C8 aggregate verifier steps/s | 308.74 | 310.74 | +0.65% | +| Cold 32K prefill tok/s | 16,872 | 16,694 | −1.05% | + +Decode uses llm-decode-bench 0.4.29, zero initial context, a 15-second warmup, +one 30-second cell per concurrency and an 8192-output-token request limit. +Both arms have zero API errors. C8 emitted tokens per verifier step change +2.603 → 2.552; output throughput alone is not a pure execution-speed metric. + +Prefill measures exact 32,768-token, one-output-token requests for 30 seconds +after warmup. All 17 reference and 16 candidate samples report 32,768 locally +computed tokens and no cache or external-transfer tokens. Reference range: +16,423–16,956 tok/s; candidate: 16,651–16,768. + +Conclusion: these observations meet a bounded 2% screening threshold. One +decode cell per arm does not establish statistical equivalence, a general +speedup or performance at other modes/concurrencies. + +## Correctness evidence and limits + +- Final vLLM composition: 393 CPU tests pass for fairness, boundary admission, + sparse cleanup, connector configuration and native-import isolation. + Recipe/composer suites: 199 passed. +- Composed GLM indexer/checkpoint GPU suites: 73 passed, including restore + byte offsets above 2 GiB. +- B12X policy/scratch/scale/alias: 70 passed. Exact GLM M8 graph oracle and + eight MHC specialization cases pass. Six two-GPU PCIe PDL torture cases + pass, including epoch rollover and multiple streams. +- LMCache filesystem/native-adapter tests: 116 passed. Its complete source + tree is identical to R34, not merely patch-equivalent. +- Packaged DFlash: three identical 8192-token requests restore exactly 8192 + tokens and emit identical greedy token IDs. Response continuation matches + an independently cold control. Shared SYSTEM reuse, C4 document lookups, + assistant continuation, tool history and changed instruction values pass. + +The fused B12X MHC post/pre oracle has the same two-element BF16 discrepancy +on master, R34 and the composed source: absolute error 0.0078125 versus its +0.004 tolerance. No tolerance was weakened. This discrepancy is not fixed or +introduced by the composition. + +No fresh no-spec/MTP3 performance matrix, DCP4 external-cache restart matrix, +TP8, NVFP4 target-KV, Qwen or DeepSeek serving qualification is claimed. +The strict-JSON/concurrent-MTP/LMCache issue tracked by vLLM #726 remains open. diff --git a/recipes/glm53/review-stack.b12x.json b/recipes/glm53/review-stack.b12x.json new file mode 100644 index 00000000..d2b20965 --- /dev/null +++ b/recipes/glm53/review-stack.b12x.json @@ -0,0 +1,23 @@ +{ + "schema_version": 1, + "name": "B12X caller-owned execution review composition", + "repository": "https://github.com/local-inference-lab/b12x.git", + "base_ref": "refs/heads/master", + "base_commit": "00b69ac22e21413622c4ecd98f607a2c3e015161", + "expected_tree": "d963f030c72a426753d2afa412b35d29eeb045f7", + "composition_strategy": "merge", + "pull_requests": [ + { + "number": 279, + "head": "637f056d2a727e732f5ec49907ef4ee4d89c43e5" + }, + { + "number": 280, + "head": "d65b1a32b2cb9022e017370d837cb4df811fdf0e" + }, + { + "number": 356, + "head": "6f356271d5b697f25b3bbf1586b198d4f9747c57" + } + ] +} diff --git a/recipes/glm53/review-stack.lmcache.json b/recipes/glm53/review-stack.lmcache.json new file mode 100644 index 00000000..f4366f47 --- /dev/null +++ b/recipes/glm53/review-stack.lmcache.json @@ -0,0 +1,59 @@ +{ + "schema_version": 1, + "name": "Engine-driven checkpoint review-tree audit", + "repository": "https://github.com/local-inference-lab/LMCache.git", + "base_ref": "refs/heads/dev", + "base_commit": "7ed4675404a31f4ffafd98975899dc83832ba965", + "expected_tree": "5a88a1ea9d2627c76288d056e7193f7454669b64", + "composition_strategy": "merge", + "pull_requests": [ + { + "number": 49, + "head": "b4b79a0ffeeec0e7ede02744cb50a8271bb1c89e" + }, + { + "number": 50, + "head": "2151e373c4fa954edeb3b170c3ce9c77550c00b4", + "requires": [ + 49 + ] + }, + { + "number": 51, + "head": "63919a2c6c310f9b34de7049b9b28b77fab13ca0", + "requires": [ + 50 + ] + }, + { + "number": 55, + "head": "078199fa3e08a0a515eae738194d29119d523b2f" + }, + { + "number": 56, + "head": "a726d4673d1f4b1219fd35e5183c9e38bd9ae119" + }, + { + "number": 61, + "head": "e0314df904cc61e988e81d8fb8b495a5ef1c272d" + }, + { + "number": 62, + "head": "91bbbb837cfe4e8c4360f3d61936c8401eabfeff" + }, + { + "number": 65, + "head": "da7df9299ee4d02d9101b18259ffc96b65792d18", + "requires": [ + 62 + ] + }, + { + "number": 66, + "head": "d59dc499b5b1b8f60716b5e6a671faa20356b42b", + "requires": [ + 62 + ] + } + ] +} diff --git a/recipes/glm53/review-stack.vllm.json b/recipes/glm53/review-stack.vllm.json new file mode 100644 index 00000000..c88ce909 --- /dev/null +++ b/recipes/glm53/review-stack.vllm.json @@ -0,0 +1,146 @@ +{ + "schema_version": 1, + "name": "Jovian serving review-tree audit", + "repository": "https://github.com/local-inference-lab/vllm.git", + "base_ref": "refs/heads/dev/jovian-judgement", + "base_commit": "a6571d0a602ed42525a7a77cf960f39ca7f8d7a8", + "expected_tree": "6d5cc3490735a53e849a4f0909183cb5e3a22e8b", + "composition_strategy": "merge", + "pull_requests": [ + { + "number": 728, + "head": "ef94f0c49d58a437089d7c4b939f695934673157" + }, + { + "number": 552, + "head": "0b33c223e62635914cba92136a385bc039114eb9" + }, + { + "number": 558, + "head": "5588b93f754a7f399106325add6881b6b02f388a" + }, + { + "number": 619, + "head": "f4516f7d35db80a6ec9d7da59ff2ba4c964aaa37" + }, + { + "number": 561, + "head": "8000b0545a33f86ee03aaa3c098039beaa5e9593" + }, + { + "number": 599, + "head": "c2a13c2c0c000580f0619304ec8bae8767ab6541" + }, + { + "number": 665, + "head": "2a8d36bb1d2e40c0f9011f9259ddcd08fe55df1d" + }, + { + "number": 666, + "head": "97dab213776f115d55642a79b64a4b002ce47a90" + }, + { + "number": 573, + "head": "cfa34ed7770512640fef3bdc219594cb8e07ac00" + }, + { + "number": 571, + "head": "52d2eca7a29d85f65b6db2aef7a3b5771353fc13" + }, + { + "number": 572, + "head": "bede2cc2a68ecd97f5067f67ca5e95980f18a80c" + }, + { + "number": 653, + "head": "44e6766e3397e8fe8ed9c1fa8a8d2783bb4a2ae8" + }, + { + "number": 660, + "head": "3bf2e1973d8f6d6b16e0a777880f46dda5b1a5a7" + }, + { + "number": 664, + "head": "c498799c0e57f1b4b0992aea935703c417685e70" + }, + { + "number": 676, + "head": "74cf9209092b3b8d986aecd0d55983ed6f9d8724" + }, + { + "number": 553, + "head": "c6b1530e895e5188b6f57109db1f609f5615ce19" + }, + { + "number": 708, + "head": "e1013ed7e15b5c845075384cc55052c16100b49b" + }, + { + "number": 709, + "head": "c57dbd68a3b6e48aabc866039d1b16a1761d56d0", + "requires": [ + 664, + 553 + ] + }, + { + "number": 706, + "head": "0627ffd88559b4cd8bd6b7672939d3ccc55899ac" + }, + { + "number": 707, + "head": "1011bfcedb1db06f97349afcc13a72c4df0bbf2f" + }, + { + "number": 710, + "head": "0eff58055a9c193f98c132b9528b057a244e32a7" + }, + { + "number": 701, + "head": "277bd8b5c2db6f650668042798e905c81000f936" + }, + { + "number": 628, + "head": "56e1bf7ccb5ea8d29e70b9ee4387cea433cf7b14" + }, + { + "number": 630, + "head": "5b6fb80f5c868b62da2c01c3f52861b34c84d8ac" + }, + { + "number": 671, + "head": "6efc1168a65b68b1ef2aa566d044d4ed2f3ef3df" + }, + { + "number": 679, + "head": "423a8653f27d680c0c7254ea8406cf75537d20d5" + }, + { + "number": 720, + "head": "fe9acad0bb6966b30e8925c6eb14386b3b2d3dbf" + }, + { + "number": 715, + "head": "618562444d73742e4e872defcad4d37f477f5c59" + }, + { + "number": 718, + "head": "25fc3585faee243b345dce770a3f1e0ee59a0ee2" + }, + { + "number": 729, + "head": "d261db0e608406e32b3cd6018a5b3ce1fbad3518", + "requires": [ + 718 + ] + }, + { + "number": 724, + "head": "7b8fe75ec529166c0d8359eed5363117c3cb3627" + }, + { + "number": 730, + "head": "ec2b235126d45153f8b9a133460a9de414fb895f" + } + ] +} diff --git a/recipes/glm53/serve-ds4-jovian.sh b/recipes/glm53/serve-ds4-jovian.sh new file mode 100644 index 00000000..bda660a0 --- /dev/null +++ b/recipes/glm53/serve-ds4-jovian.sh @@ -0,0 +1,59 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Model-specific entrypoint over the same installed vLLM/B12X/LMCache packages +# as the GLM launcher. DS4_* controls isolate graph and CPU settings from the +# GLM defaults embedded in the image; native serving options remain available. +export PATH="/opt/venv/bin:${PATH}" +export PYTHON_BIN=${PYTHON_BIN:-/opt/venv/bin/python} +export DS4_MODEL_VARIANT=${DS4_MODEL_VARIANT:-text} +case "${DS4_MODEL_VARIANT}" in + text) default_seqs=8; default_drafts=5 ;; + vision) default_seqs=4; default_drafts=3 ;; + *) echo 'DS4_MODEL_VARIANT must be text or vision.' >&2; exit 2 ;; +esac +export MODE=${MODE:-dspark} BACKEND=${BACKEND:-b12x-a8-dglin} +export TP_SIZE=${TP_SIZE:-${TP:-2}} DCP_SIZE=${DCP_SIZE:-${DCP:-1}} +export DSPARK_DEPTH_MODE=${DSPARK_DEPTH_MODE:-fixed} +export DSPARK_TOKENS=${DSPARK_TOKENS:-${default_drafts}} +export DRAFT_SAMPLE_METHOD=${DRAFT_SAMPLE_METHOD:-probabilistic} +export REJECTION_SAMPLE_METHOD=${REJECTION_SAMPLE_METHOD:-standard} +export MAX_MODEL_LEN=${MAX_MODEL_LEN:--1} +export MAX_NUM_SEQS=${MAX_NUM_SEQS:-${default_seqs}} +export MAX_NUM_BATCHED_TOKENS=${MAX_NUM_BATCHED_TOKENS:-4096} +export OMP_NUM_THREADS=${DS4_OMP_NUM_THREADS:-2} +export MAX_CUDAGRAPH_CAPTURE_SIZE=${DS4_MAX_CUDAGRAPH_CAPTURE_SIZE:-auto} +export CUDAGRAPH_CAPTURE_SIZES=${DS4_CUDAGRAPH_CAPTURE_SIZES:-default} +export LOAD_FORMAT=${LOAD_FORMAT:-instanttensor} +export INSTANTTENSOR_BACKEND=${INSTANTTENSOR_BACKEND:-BUFFERED} +export LMCACHE_MODE=${LMCACHE_MODE:-off} +export LMCACHE_AUTO_TRANSFER_MODE=${LMCACHE_AUTO_TRANSFER_MODE:-engine_driven} +export LMCACHE_CHUNK_SIZE=${LMCACHE_CHUNK_SIZE:-4096} +export LMCACHE_SEPARATE_OBJECT_GROUPS=${LMCACHE_SEPARATE_OBJECT_GROUPS:-1} +export LMCACHE_L1_GB=${LMCACHE_L1_GB:-24} +unset NCCL_GRAPH_FILE + +generation_defaults_from_cli=0 +for argument in "$@"; do + case "${argument//_/-}" in + --generation-config | --generation-config=* | \ + --override-generation-config | --override-generation-config=* | \ + --override-generation-config.* | --config | --config=*) + generation_defaults_from_cli=1 ;; + esac +done +# The native DS4 launcher also accepts whitespace-separated EXTRA_VLLM_ARGS. +# Treat a generation configuration there as explicit operator policy. +extra_options=${EXTRA_VLLM_ARGS:-} +case " ${extra_options//_/-} " in + *' --generation-config'* | *' --override-generation-config'* | *' --config'*) + generation_defaults_from_cli=1 ;; +esac +if ((generation_defaults_from_cli == 0)); then + # DeepSeek recommends these values for agentic text and Vision use. Native + # request parameters take precedence over the server's sampling defaults. + set -- --override-generation-config '{"temperature":1.0,"top_p":0.95}' "$@" +fi + +exec /usr/local/bin/lmcache-mp-wrapper.sh \ + /opt/glm53-flash/vllm/serve-ds4-flash.sh "$@" diff --git a/recipes/glm53/serve-glm53-flash-cache-complete.sh b/recipes/glm53/serve-glm53-flash-cache-complete.sh new file mode 100644 index 00000000..4e1af47e --- /dev/null +++ b/recipes/glm53/serve-glm53-flash-cache-complete.sh @@ -0,0 +1,413 @@ +#!/usr/bin/env bash +set -euo pipefail + +readonly cache_launcher=/usr/local/libexec/serve-glm53-flash-lmcache-cache-complete.sh + +if (($# == 1)) && [[ $1 == --help || $1 == -h ]]; then + exec /usr/local/bin/serve-glm53-flash-nvfp4-dflash2.sh "$@" +fi + +# Explicit cache policy or geometry also selects the cache-mode contract. With +# no cache-related setting, preserve the image's ordinary serving command. +explicit_checkpoint_policy=0 +for arg in "$@"; do + case "${arg}" in + --recurrent-checkpoint-policy | --recurrent-checkpoint-policy=*) + explicit_checkpoint_policy=1 + ;; + esac +done +if [[ -z ${CACHE_MODE+x} && -z ${LMCACHE_ENABLED+x} && \ + -z ${GLM53_TARGET_BLOCK_SIZE+x} && -z ${GLM53_MAMBA_BLOCK_SIZE+x} && \ + ${explicit_checkpoint_policy} == 0 && \ + ${CACHE_CONFIG_DRY_RUN:-0} != 1 ]]; then + exec "${cache_launcher}" "$@" +fi + +fail() { + printf '%s\n' "$1" >&2 + exit 2 +} + +require_positive_integer() { + local name=$1 + local value=$2 + [[ ${value} =~ ^[1-9][0-9]*$ ]] || + fail "${name} must be a positive integer; got ${value}" +} + +require_nonnegative_integer() { + local name=$1 + local value=$2 + [[ ${value} =~ ^[0-9]+$ ]] || + fail "${name} must be a non-negative integer; got ${value}" +} + +sanitize_namespace_component() { + local value=$1 + value=${value//[^A-Za-z0-9._-]/_} + [[ -n ${value} ]] || value=unset + printf '%s' "${value}" +} + +# Reject command-line values for settings owned by this launcher's environment +# contract. Emitting a setting twice lets argument order silently select a cache +# layout that does not match the persistent-object namespace. +for arg in "$@"; do + case "${arg}" in + --kv-cache-dtype | --kv-cache-dtype=* | \ + --kv-offloading-backend | --kv-offloading-backend=* | \ + --kv-offloading-size | --kv-offloading-size=* | \ + --kv-transfer-config | --kv-transfer-config=* | \ + --enable-cumem-allocator) + fail "${arg%%=*} is managed by CACHE_MODE and the cache environment variables" + ;; + esac +done + +cache_mode=${CACHE_MODE:-} +if [[ -z ${cache_mode} ]]; then + if [[ ${LMCACHE_ENABLED:-0} == 1 ]]; then + cache_mode=lmcache + else + cache_mode=vram + fi +fi +case "${cache_mode}" in + vram | native | lmcache) ;; + *) fail "CACHE_MODE must be vram, native, or lmcache; got ${cache_mode}" ;; +esac + +# One policy selects both recurrent retention and its compatible external +# connector. A second launcher-specific policy could disagree with vLLM. +checkpoint_policy=auto +policy_value_pending=0 +policy_seen=0 +forwarded_args=() +for arg in "$@"; do + if ((policy_value_pending)); then + checkpoint_policy=${arg} + policy_value_pending=0 + continue + fi + case "${arg}" in + --recurrent-checkpoint-policy | --recurrent-checkpoint-policy=*) + ((policy_seen == 0)) || fail 'Specify --recurrent-checkpoint-policy only once' + policy_seen=1 + if [[ ${arg} == *=* ]]; then + checkpoint_policy=${arg#*=} + else + policy_value_pending=1 + fi + ;; + *) forwarded_args+=("${arg}") ;; + esac +done +((policy_value_pending == 0)) || fail '--recurrent-checkpoint-policy requires a value' +case "${checkpoint_policy}" in + auto | aligned | request_boundaries) ;; + *) fail "Unsupported recurrent checkpoint policy: ${checkpoint_policy}" ;; +esac +set -- "${forwarded_args[@]}" +checkpoint_model=${MODEL:-local-inference-lab/GLM-5.3-Flash-NVFP4} +if (($# > 0)) && [[ $1 != -* ]]; then + checkpoint_model=$1 +fi + +# External-cache object geometry owns retention boundaries. GPU-local caching +# has no transfer-object constraint and accepts vLLM's native retention option. +if [[ ${cache_mode} != vram ]]; then + for arg in "$@"; do + case "${arg}" in + --prefix-cache-retention-interval | --prefix-cache-retention-interval=*) + fail "${arg%%=*} is managed by CACHE_MODE and the cache environment variables" + ;; + esac + done +fi + +tp=${TP:-4} +dcp=${DCP:-1} +require_positive_integer TP "${tp}" +require_positive_integer DCP "${dcp}" +case "${tp}" in + 4 | 8) ;; + *) fail "The cache-complete launcher supports TP=4 or TP=8; got ${tp}" ;; +esac +if ((tp % dcp != 0)); then + fail "DCP must divide TP; got TP=${tp} DCP=${dcp}" +fi + +interleave=${CP_KV_CACHE_INTERLEAVE_SIZE:-4} +require_positive_integer CP_KV_CACHE_INTERLEAVE_SIZE "${interleave}" +gather_selector=${DCP_CKV_GATHER:-auto} +case "${gather_selector}" in + auto) + if ((dcp > 1)); then + resolved_gather=1 + else + resolved_gather=0 + fi + ;; + 0 | 1) resolved_gather=${DCP_CKV_GATHER} ;; + *) fail "DCP_CKV_GATHER must be auto, 0, or 1; got ${gather_selector}" ;; +esac + +kv_cache_quant=${KV_CACHE_QUANT:-} +if [[ -z ${kv_cache_quant} ]]; then + configured_kv_cache_dtype=${KV_CACHE_DTYPE:-fp8} + case "${configured_kv_cache_dtype}" in + fp8 | fp8_e4m3 | fp8_ds_mla) kv_cache_quant=fp8_ds_mla ;; + nvfp4_ds_mla) kv_cache_quant=nvfp4_ds_mla ;; + *) + fail "KV_CACHE_DTYPE is not representable by KV_CACHE_QUANT: ${configured_kv_cache_dtype}" + ;; + esac +fi +case "${kv_cache_quant}" in + fp8_ds_mla) + vllm_kv_cache_dtype=fp8 + lmcache_kv_cache_dtype=fp8_ds_mla + ;; + nvfp4_ds_mla) + vllm_kv_cache_dtype=nvfp4_ds_mla + lmcache_kv_cache_dtype=nvfp4_ds_mla + ;; + *) + fail "KV_CACHE_QUANT must be fp8_ds_mla or nvfp4_ds_mla; got ${kv_cache_quant}" + ;; +esac + +# LMCache objects are global sequence ranges. DCP shards each target cache +# block across ranks, so automatic geometry resolves one complete per-rank +# page inside the configured retention interval. GPU-only and native-offload +# modes use the capacity-efficient 2,048-token page unless overridden. +if [[ -n ${GLM53_TARGET_BLOCK_SIZE:-} ]]; then + target_block_size=${GLM53_TARGET_BLOCK_SIZE} +elif [[ ${cache_mode} == lmcache ]]; then + target_block_size=auto +else + target_block_size=2048 +fi +mamba_block_size=${GLM53_MAMBA_BLOCK_SIZE:-auto} +if [[ ${target_block_size} != auto ]]; then + require_positive_integer GLM53_TARGET_BLOCK_SIZE "${target_block_size}" + ((target_block_size % 64 == 0)) || + fail "GLM53_TARGET_BLOCK_SIZE must be a multiple of 64; got ${target_block_size}" +fi +if [[ ${mamba_block_size} != auto ]]; then + require_positive_integer GLM53_MAMBA_BLOCK_SIZE "${mamba_block_size}" + ((mamba_block_size % 64 == 0)) || + fail "GLM53_MAMBA_BLOCK_SIZE must be a multiple of 64; got ${mamba_block_size}" + if [[ ${target_block_size} != auto ]] && + ((mamba_block_size % target_block_size != 0 && + target_block_size % mamba_block_size != 0)); then + fail "GLM53_MAMBA_BLOCK_SIZE must be a multiple or divisor of GLM53_TARGET_BLOCK_SIZE" + fi +fi + +speculator=${SPECULATOR:-mtp} +case "${speculator}" in + mtp) + spec_depth=${MTP_DEPTH:-${MTP:-${NUM_SPECULATIVE_TOKENS:-0}}} + require_nonnegative_integer MTP_DEPTH "${spec_depth}" + ((spec_depth <= 5)) || fail "MTP_DEPTH must be between 0 and 5; got ${spec_depth}" + ;; + dflash | dflash2) + speculator=dflash2 + spec_depth=${DFLASH_DEPTH:-${NUM_SPECULATIVE_TOKENS:-7}} + require_positive_integer DFLASH_DEPTH "${spec_depth}" + ((spec_depth <= 7)) || + fail "DFLASH_DEPTH must be between 1 and 7; got ${spec_depth}" + ;; + *) fail "SPECULATOR must be mtp or dflash2; got ${speculator}" ;; +esac + +export TP=${tp} +export DCP=${dcp} +export SPECULATOR=${speculator} +export NUM_SPECULATIVE_TOKENS=${spec_depth} +export KV_CACHE_DTYPE=${vllm_kv_cache_dtype} +export LMCACHE_VLLM_KV_CACHE_DTYPE=${vllm_kv_cache_dtype} +export LMCACHE_KV_CACHE_DTYPE=${lmcache_kv_cache_dtype} +export VLLM_GLM53_SPLIT_TARGET_BLOCK_SIZE=${target_block_size} +export VLLM_GLM53_SPLIT_MAMBA_BLOCK_SIZE=${mamba_block_size} + +case "${cache_mode}" in + vram) + export LMCACHE_ENABLED=0 + if [[ ${checkpoint_policy} == aligned ]]; then + retention_supplied=0 + for arg in "$@"; do + case "${arg}" in + --prefix-cache-retention-interval | --prefix-cache-retention-interval=*) + retention_supplied=1 + ;; + esac + done + # Packed native exports retain interior recurrent states without making + # each checkpoint a separate scheduler forward. Explicit retention limits + # remain available to deployments that prefer fewer cached states. + if ((retention_supplied == 0)); then + set -- "$@" --prefix-cache-retention-interval None + fi + fi + ;; + native) + export LMCACHE_ENABLED=0 + native_offload_size=${NATIVE_KV_OFFLOADING_SIZE_GB:-64} + [[ ${native_offload_size} =~ ^[0-9]+([.][0-9]+)?$ ]] || + fail "NATIVE_KV_OFFLOADING_SIZE_GB must be a positive number; got ${native_offload_size}" + [[ ! ${native_offload_size} =~ ^0+([.]0+)?$ ]] || + fail "NATIVE_KV_OFFLOADING_SIZE_GB must be greater than zero" + set -- "$@" \ + --kv-offloading-backend native \ + --kv-offloading-size "${native_offload_size}" \ + --enable-cumem-allocator + ;; + lmcache) + export LMCACHE_ENABLED=1 + lmcache_transfer_mode=${LMCACHE_TRANSFER_MODE:-engine_driven} + case "${lmcache_transfer_mode}" in + engine_driven | lmcache_driven | auto) ;; + *) + fail "LMCACHE_TRANSFER_MODE must be engine_driven, lmcache_driven, or auto; got ${lmcache_transfer_mode}" + ;; + esac + export LMCACHE_TRANSFER_MODE=${lmcache_transfer_mode} + if [[ ${checkpoint_policy} == auto ]]; then + if [[ ${lmcache_transfer_mode} == engine_driven ]]; then + checkpoint_policy=request_boundaries + else + checkpoint_policy=aligned + fi + fi + if [[ ${checkpoint_policy} == aligned && -n ${LMCACHE_CHECKPOINT_IDENTITY:-} ]]; then + fail 'An immutable semantic checkpoint identity is incompatible with aligned transfers' + fi + if [[ ${checkpoint_policy} == request_boundaries ]]; then + [[ ${lmcache_transfer_mode} == engine_driven ]] || + fail 'Request-boundary LMCache requires engine_driven transfer' + fi + # Both external formats need immutable weight identities. Transfer policy + # changes the connector, not whether filesystem payloads can be reused. + if [[ ${checkpoint_policy} == request_boundaries || ${LMCACHE_L2_ENABLED:-1} == 1 ]]; then + for arg in "$@"; do + case "${arg}" in + --revision | --revision=* | --speculative-config | --speculative-config=*) + fail "${arg%%=*} would override the authenticated checkpoint; use MODEL_REVISION and the speculative-mode environment settings" + ;; + esac + done + resolved_checkpoint_identity=${LMCACHE_CHECKPOINT_IDENTITY:-} + if [[ -z ${resolved_checkpoint_identity} && ${CACHE_CONFIG_DRY_RUN:-0} != 1 ]]; then + identity_speculation=none + if ((spec_depth > 0)); then + identity_speculation=mtp + [[ ${speculator} != dflash2 ]] || identity_speculation=dflash + fi + identity_result=$(/opt/venv/bin/python \ + /usr/local/libexec/glm53_checkpoint_identity.py \ + --model "${checkpoint_model}" --revision "${MODEL_REVISION:-}" \ + --draft-model "${DFLASH_MODEL:-local-inference-lab/GLM-5.3-Flash-DFlash2}" \ + --draft-revision "${DFLASH_MODEL_REVISION:-}" \ + --speculation "${identity_speculation}") + resolved_checkpoint_identity=$(jq -cer '.checkpoint_identity' <<< "${identity_result}") + MODEL_REVISION=$(jq -er '.model_revision' <<< "${identity_result}") + DFLASH_MODEL_REVISION=$(jq -er '.draft_model_revision' <<< "${identity_result}") + export MODEL_REVISION DFLASH_MODEL_REVISION + fi + if [[ -n ${resolved_checkpoint_identity} ]]; then + jq -e --argjson depth "${spec_depth}" ' + type == "object" and + (.target_revision | type == "string" and test("^([0-9a-f]{40}|[0-9a-f]{64})$")) and + (.source_revision | type == "string" and test("^([0-9a-f]{40}|[0-9a-f]{64})$")) and + (.draft_revision | type == "string" and + (test("^([0-9a-f]{40}|[0-9a-f]{64})$") or ($depth == 0 and . == ""))) + ' <<< "${resolved_checkpoint_identity}" >/dev/null || + fail 'Persistent checkpoint identity requires immutable target, source and active draft revisions' + LMCACHE_MODEL_REVISION_ID=$(jq -er '.target_revision' <<< "${resolved_checkpoint_identity}") + LMCACHE_DFLASH_REVISION_ID=$(jq -er '.draft_revision' <<< "${resolved_checkpoint_identity}") + export LMCACHE_MODEL_REVISION_ID LMCACHE_DFLASH_REVISION_ID + if [[ ${checkpoint_policy} == request_boundaries ]]; then + export LMCACHE_CHECKPOINT_IDENTITY=${resolved_checkpoint_identity} + fi + fi + fi + if [[ ${lmcache_transfer_mode} == engine_driven && \ + -z ${GPU_MEMORY_UTILIZATION+x} ]]; then + export GPU_MEMORY_UTILIZATION=0.950 + fi + lmcache_chunk_size=${LMCACHE_CHUNK_SIZE:-4096} + target_token_budget=${LMCACHE_TARGET_TOKEN_BUDGET:-4096} + require_positive_integer LMCACHE_CHUNK_SIZE "${lmcache_chunk_size}" + require_positive_integer LMCACHE_TARGET_TOKEN_BUDGET "${target_token_budget}" + if [[ ${lmcache_chunk_size} != "${target_token_budget}" ]]; then + fail "LMCACHE_CHUNK_SIZE and LMCACHE_TARGET_TOKEN_BUDGET must match; got ${lmcache_chunk_size} and ${target_token_budget}" + fi + if [[ ${target_block_size} != auto ]] && + ((lmcache_chunk_size % (target_block_size * dcp) != 0)); then + fail "LMCACHE_CHUNK_SIZE must be a multiple of GLM53_TARGET_BLOCK_SIZE times DCP; got chunk ${lmcache_chunk_size}, block ${target_block_size}, and DCP ${dcp}" + fi + export LMCACHE_CHUNK_SIZE=${lmcache_chunk_size} + export LMCACHE_TARGET_TOKEN_BUDGET=${target_token_budget} + + case "${LMCACHE_L2_ENABLED:-1}" in + 0 | 1) ;; + *) fail "LMCACHE_L2_ENABLED must be 0 or 1; got ${LMCACHE_L2_ENABLED}" ;; + esac + if [[ ${LMCACHE_L2_ENABLED:-1} == 1 ]]; then + l2_root=${LMCACHE_L2_ROOT:-${LMCACHE_L2_PATH:-/lmcache-l2}} + [[ ${l2_root} == /* && ${l2_root} != *\"* && \ + ${l2_root} != *\\* && ${l2_root} != *\'* ]] || + fail "LMCACHE_L2_ROOT must be an absolute path without quotes or backslashes" + + schema=$(sanitize_namespace_component "${LMCACHE_SCHEMA_REVISION:-glm53-r18-cache-v1}") + model_id=$(sanitize_namespace_component "${checkpoint_model}") + # Dry-run deliberately performs no model I/O and must not claim identity. + model_revision=$(sanitize_namespace_component "${LMCACHE_MODEL_REVISION_ID:-unresolved-dry-run}") + draft_revision=none + if [[ ${speculator} == dflash2 ]]; then + draft_revision=$(sanitize_namespace_component "${LMCACHE_DFLASH_REVISION_ID:-unresolved-dry-run}") + fi + namespace="${schema}/${model_id}-${model_revision}/kv-${kv_cache_quant}/tp-${tp}-dcp-${dcp}-i-${interleave}-g-${resolved_gather}/spec-${speculator}-${spec_depth}-draft-${draft_revision}/chunk-${lmcache_chunk_size}" + export LMCACHE_L2_PATH="${l2_root%/}/${namespace}" + export LMCACHE_INSTANCE_ID=${LMCACHE_INSTANCE_ID:-"glm53-${schema}-tp${tp}-dcp${dcp}-${kv_cache_quant}"} + if [[ ${checkpoint_policy} == request_boundaries ]]; then + export LMCACHE_CHECKPOINT_INDEX_PATH=${LMCACHE_CHECKPOINT_INDEX_PATH:-"${LMCACHE_L2_PATH}/semantic-directory.sqlite3"} + fi + fi + ;; +esac + +if [[ ${cache_mode} == lmcache ]] || ((policy_seen)); then + set -- "$@" --recurrent-checkpoint-policy "${checkpoint_policy}" +fi + +if [[ ${CACHE_CONFIG_DRY_RUN:-0} == 1 ]]; then + printf 'CACHE_MODE=%q\n' "${cache_mode}" + printf 'KV_CACHE_QUANT=%q\n' "${kv_cache_quant}" + printf 'TP=%q\nDCP=%q\nDCP_CKV_GATHER=%q\n' \ + "${tp}" "${dcp}" "${resolved_gather}" + printf 'SPECULATOR=%q\nSPECULATIVE_DEPTH=%q\n' \ + "${speculator}" "${spec_depth}" + printf 'VLLM_GLM53_SPLIT_TARGET_BLOCK_SIZE=%q\n' \ + "${VLLM_GLM53_SPLIT_TARGET_BLOCK_SIZE}" + printf 'VLLM_GLM53_SPLIT_MAMBA_BLOCK_SIZE=%q\n' \ + "${VLLM_GLM53_SPLIT_MAMBA_BLOCK_SIZE}" + if [[ ${cache_mode} == lmcache ]]; then + printf 'RECURRENT_CHECKPOINT_POLICY=%q\n' "${checkpoint_policy}" + printf 'LMCACHE_TRANSFER_MODE=%q\n' "${LMCACHE_TRANSFER_MODE}" + printf 'GPU_MEMORY_UTILIZATION=%q\n' \ + "${GPU_MEMORY_UTILIZATION:-launcher-default}" + printf 'LMCACHE_L2_PATH=%q\n' "${LMCACHE_L2_PATH:-disabled}" + fi + printf 'ARGV:' + if (($# > 0)); then + printf ' %q' "$@" + fi + printf '\n' + exit 0 +fi + +exec "${cache_launcher}" "$@" diff --git a/recipes/glm53/serve-glm53-flash-lmcache-cache-complete.sh b/recipes/glm53/serve-glm53-flash-lmcache-cache-complete.sh new file mode 100644 index 00000000..93358bc8 --- /dev/null +++ b/recipes/glm53/serve-glm53-flash-lmcache-cache-complete.sh @@ -0,0 +1,481 @@ +#!/usr/bin/env bash +set -euo pipefail + +readonly base_launcher=/usr/local/bin/serve-glm53-flash-nvfp4-dflash2.sh + +# LMCache is opt-in so the default entrypoint preserves the qualified GLM-5.3 +# serving command, allocator, scheduler, and backend configuration exactly. +if [[ ${LMCACHE_ENABLED:-0} == 0 ]]; then + exec "${base_launcher}" "$@" +fi +if [[ ${LMCACHE_ENABLED} != 1 ]]; then + printf 'LMCACHE_ENABLED must be 0 or 1; got %s\n' "${LMCACHE_ENABLED}" >&2 + exit 2 +fi + +require_positive_integer() { + local name=$1 + local value=$2 + if [[ ! ${value} =~ ^[1-9][0-9]*$ ]]; then + printf '%s must be a positive integer; got %s\n' "${name}" "${value}" >&2 + exit 2 + fi +} + +readonly mp_host=${LMCACHE_MP_HOST:-127.0.0.1} +readonly http_host=${LMCACHE_HTTP_HOST:-127.0.0.1} +readonly mp_port=${LMCACHE_MP_PORT:-5555} +readonly http_port=${LMCACHE_HTTP_PORT:-8085} +readonly prometheus_port=${LMCACHE_PROMETHEUS_PORT:-9095} +readonly startup_timeout=${LMCACHE_STARTUP_TIMEOUT_SECONDS:-120} +readonly transfer_mode=${LMCACHE_TRANSFER_MODE:-engine_driven} +readonly broker_dir=${LMCACHE_CUMEM_BROKER_DIR:-/cache/lmcache-cumem} +readonly prefetch_policy=${LMCACHE_L2_PREFETCH_POLICY:-retain} +readonly server_extra_args=${LMCACHE_SERVER_EXTRA_ARGS:-} +readonly chunk_size=${LMCACHE_CHUNK_SIZE:-4096} +readonly target_token_budget=${LMCACHE_TARGET_TOKEN_BUDGET:-${MAX_NUM_BATCHED_TOKENS:-4096}} +readonly max_num_seqs=${MAX_NUM_SEQS:-16} +readonly min_shm_gib=${LMCACHE_MIN_SHM_GIB:-96} +readonly lmcache_kv_cache_dtype=${LMCACHE_KV_CACHE_DTYPE:-nvfp4_ds_mla} +readonly vllm_kv_cache_dtype=${LMCACHE_VLLM_KV_CACHE_DTYPE:-${lmcache_kv_cache_dtype}} +readonly instance_id=${LMCACHE_INSTANCE_ID:-glm53-jovian-judgement-lmcache} +readonly checkpoint_identity=${LMCACHE_CHECKPOINT_IDENTITY:-} +checkpoint_policy_supplied=0 +checkpoint_policy_pending=0 +checkpoint_cli_policy=auto +for arg in "$@"; do + if ((checkpoint_policy_pending)); then + checkpoint_cli_policy=${arg} + checkpoint_policy_pending=0 + continue + fi + case "${arg}" in + --recurrent-checkpoint-policy | --recurrent-checkpoint-policy=*) + if ((checkpoint_policy_supplied)); then + printf 'Specify --recurrent-checkpoint-policy only once\n' >&2 + exit 2 + fi + checkpoint_policy_supplied=1 + if [[ ${arg} == *=* ]]; then + checkpoint_cli_policy=${arg#*=} + else + checkpoint_policy_pending=1 + fi + ;; + esac +done +if ((checkpoint_policy_pending)); then + printf '%s\n' '--recurrent-checkpoint-policy requires a value' >&2 + exit 2 +fi +if [[ -n ${checkpoint_identity} && ${checkpoint_cli_policy} != auto && + ${checkpoint_cli_policy} != request_boundaries ]]; then + printf 'Semantic checkpoints require request_boundaries or auto\n' >&2 + exit 2 +fi + +require_positive_integer LMCACHE_MP_PORT "${mp_port}" +require_positive_integer LMCACHE_HTTP_PORT "${http_port}" +require_positive_integer LMCACHE_PROMETHEUS_PORT "${prometheus_port}" +require_positive_integer LMCACHE_STARTUP_TIMEOUT_SECONDS "${startup_timeout}" +require_positive_integer LMCACHE_CHUNK_SIZE "${chunk_size}" +require_positive_integer LMCACHE_TARGET_TOKEN_BUDGET "${target_token_budget}" +require_positive_integer MAX_NUM_SEQS "${max_num_seqs}" +require_positive_integer LMCACHE_MIN_SHM_GIB "${min_shm_gib}" +if [[ ! ${mp_host} =~ ^[A-Za-z0-9_.:-]+$ ]]; then + printf 'LMCACHE_MP_HOST contains unsupported characters: %s\n' \ + "${mp_host}" >&2 + exit 2 +fi +if [[ ! ${http_host} =~ ^[A-Za-z0-9_.:][A-Za-z0-9_.:-]*$ ]]; then + printf 'LMCACHE_HTTP_HOST contains unsupported characters: %s\n' \ + "${http_host}" >&2 + exit 2 +fi +# A wildcard bind needs a concrete loopback address for the readiness probe. +http_probe_host=${http_host} +case "${http_host}" in + 0.0.0.0) http_probe_host=127.0.0.1 ;; + ::) http_probe_host=::1 ;; +esac +if [[ ${http_probe_host} == *:* ]]; then + http_probe_host="[${http_probe_host}]" +fi +readonly health_url="http://${http_probe_host}:${http_port}/healthcheck" +http_loopback=0 +if [[ ${http_host} == ::1 || ${http_host} == localhost ]]; then + http_loopback=1 +elif [[ ${http_host} =~ ^127\.([0-9]{1,3})\.([0-9]{1,3})\.([0-9]{1,3})$ ]] && + ((10#${BASH_REMATCH[1]} <= 255 && 10#${BASH_REMATCH[2]} <= 255 && + 10#${BASH_REMATCH[3]} <= 255)); then + http_loopback=1 +fi +if ((http_loopback == 0)); then + printf '%s\n' 'LMCache HTTP includes administrative APIs; restrict access to a trusted network or authenticated proxy.' >&2 +fi + +case "${transfer_mode}" in + lmcache_driven | auto | engine_driven) ;; + *) + printf 'LMCACHE_TRANSFER_MODE must be lmcache_driven, auto, or engine_driven; got %s\n' \ + "${transfer_mode}" >&2 + exit 2 + ;; +esac + +shm_name=${LMCACHE_SHM_NAME:-} +if [[ ${transfer_mode} == engine_driven && -z ${shm_name} ]]; then + # The MP port is already exclusive under the host-network serving contract, + # so including it also gives each concurrently runnable sidecar a distinct + # POSIX shared-memory arena. + safe_instance_id=${instance_id//[^A-Za-z0-9._-]/_} + shm_name="lmcache-${safe_instance_id}-${mp_port}" +fi +readonly shm_name +if [[ -n ${shm_name} && ! ${shm_name} =~ ^[A-Za-z0-9._-]+$ ]]; then + printf 'LMCACHE_SHM_NAME contains unsupported characters: %s\n' \ + "${shm_name}" >&2 + exit 2 +fi + +case "${prefetch_policy}" in + default | retain) ;; + *) + printf 'LMCACHE_L2_PREFETCH_POLICY must be default or retain; got %s\n' \ + "${prefetch_policy}" >&2 + exit 2 + ;; +esac + +case "${lmcache_kv_cache_dtype}" in + fp8_ds_mla | nvfp4_ds_mla) ;; + *) + printf 'LMCACHE_KV_CACHE_DTYPE must be fp8_ds_mla or nvfp4_ds_mla; got %s\n' \ + "${lmcache_kv_cache_dtype}" >&2 + exit 2 + ;; +esac +case "${vllm_kv_cache_dtype}" in + fp8 | nvfp4_ds_mla) ;; + *) + printf 'LMCACHE_VLLM_KV_CACHE_DTYPE must be fp8 or nvfp4_ds_mla; got %s\n' \ + "${vllm_kv_cache_dtype}" >&2 + exit 2 + ;; +esac +if [[ ${lmcache_kv_cache_dtype} == fp8_ds_mla && \ + ${vllm_kv_cache_dtype} != fp8 ]]; then + printf 'fp8_ds_mla storage requires vLLM --kv-cache-dtype fp8\n' >&2 + exit 2 +fi +if [[ ${lmcache_kv_cache_dtype} == nvfp4_ds_mla && \ + ${vllm_kv_cache_dtype} != nvfp4_ds_mla ]]; then + printf 'nvfp4_ds_mla storage requires the same vLLM cache dtype\n' >&2 + exit 2 +fi +if [[ ${chunk_size} != "${target_token_budget}" ]]; then + printf 'LMCACHE_CHUNK_SIZE and LMCACHE_TARGET_TOKEN_BUDGET must match; got %s and %s\n' \ + "${chunk_size}" "${target_token_budget}" >&2 + exit 2 +fi + +shm_bytes=$(df -B1 --output=size /dev/shm | awk 'NR == 2 {print $1}') +readonly shm_bytes +readonly min_shm_bytes=$((min_shm_gib * 1024 * 1024 * 1024)) +if [[ ! ${shm_bytes} =~ ^[0-9]+$ ]] || ((shm_bytes < min_shm_bytes)); then + printf 'LMCache native transfer requires a private /dev/shm of at least %s GiB; got %s bytes\n' \ + "${min_shm_gib}" "${shm_bytes:-unknown}" >&2 + exit 2 +fi + +lmcache_server=( + /opt/venv/bin/lmcache server + --instance-id "${instance_id}" + --host "${mp_host}" + --port "${mp_port}" + --chunk-size "${chunk_size}" + --max-workers "${LMCACHE_MAX_WORKERS:-8}" + --max-gpu-workers "${LMCACHE_MAX_GPU_WORKERS:-8}" + --max-cpu-workers "${LMCACHE_MAX_CPU_WORKERS:-16}" + --hash-algorithm blake3 + --supported-transfer-mode "${transfer_mode}" + --separate-object-groups + --l1-size-gb "${LMCACHE_L1_SIZE_GB:-64}" + --l1-init-size-gb "${LMCACHE_L1_INIT_SIZE_GB:-2}" + --eviction-policy LRU + --l2-prefetch-policy "${prefetch_policy}" + --http-host "${http_host}" + --http-port "${http_port}" + --prometheus-port "${prometheus_port}" +) + +if [[ ${transfer_mode} == engine_driven ]]; then + # Engine-driven SHM advertises direct views into the complete L1 arena. + # Lazy allocation cannot expose a stable pool address or capacity, so an + # explicit SHM name requires a preallocated L1 arena. + lmcache_server+=(--no-l1-use-lazy --shm-name "${shm_name}") +else + # Lazy L1 keeps the default pickle and CUDA-IPC transports bounded by the + # data actually stored instead of reserving the complete configured arena. + lmcache_server+=(--l1-use-lazy) +fi + +if [[ -n ${checkpoint_identity} ]]; then + if [[ ${transfer_mode} != engine_driven ]]; then + printf 'Semantic checkpoints require engine-driven SHM transport\n' >&2 + exit 2 + fi + jq -e 'type == "object"' <<< "${checkpoint_identity}" >/dev/null + if [[ -n ${LMCACHE_CHECKPOINT_INDEX_PATH:-} ]]; then + mkdir -p "$(dirname "${LMCACHE_CHECKPOINT_INDEX_PATH}")" + lmcache_server+=(--checkpoint-index-path "${LMCACHE_CHECKPOINT_INDEX_PATH}") + fi +fi + +case "${LMCACHE_L2_ENABLED:-1}" in + 0) ;; + 1) + l2_path=${LMCACHE_L2_PATH:-/lmcache-l2} + if [[ ${l2_path} != /* || ${l2_path} == *['"\']* ]]; then + printf 'LMCACHE_L2_PATH must be an absolute path without quotes or backslashes\n' >&2 + exit 2 + fi + mkdir -p "${l2_path}" + l2_config=${LMCACHE_L2_CONFIG:-} + if [[ -z ${l2_config} ]]; then + printf -v l2_config \ + '{"type":"fs_native","base_path":"%s","num_workers":%s,"use_odirect":false,"max_capacity_gb":%s,"eviction":{"eviction_policy":"LRU","trigger_watermark":0.8,"eviction_ratio":0.2}}' \ + "${l2_path}" "${LMCACHE_L2_WORKERS:-8}" \ + "${LMCACHE_L2_MAX_CAPACITY_GB:-512}" + fi + /opt/venv/bin/python -c \ + 'import json, sys; value=json.loads(sys.argv[1]); assert isinstance(value, dict)' \ + "${l2_config}" + lmcache_server+=(--l2-adapter "${l2_config}") + ;; + *) + printf 'LMCACHE_L2_ENABLED must be 0 or 1; got %s\n' \ + "${LMCACHE_L2_ENABLED}" >&2 + exit 2 + ;; +esac + +if [[ ${server_extra_args} == *$'\n'* || ${server_extra_args} == *$'\r'* ]]; then + printf 'LMCACHE_SERVER_EXTRA_ARGS must be one whitespace-separated line\n' >&2 + exit 2 +fi +if [[ ${prefetch_policy} == retain && ${LMCACHE_L2_ENABLED:-1} == 1 ]]; then + lmcache_server+=(--emergency-evict-for-prefetch) +fi +server_extra_argv=() +read -r -a server_extra_argv <<< "${server_extra_args}" +for argument in "${server_extra_argv[@]}"; do + case "${argument%%=*}" in + --instance-id | --host | --port | --http-host | --http-port | --prometheus-port | \ + --supported-transfer-mode | --chunk-size | --hash-algorithm | \ + --separate-object-groups | --no-separate-object-groups | --shm-name | \ + --l1-use-lazy | --no-l1-use-lazy | --checkpoint-index-path | --l2-prefetch-policy | \ + --emergency-evict-for-prefetch | --no-emergency-evict-for-prefetch) + printf 'LMCACHE_SERVER_EXTRA_ARGS cannot override launcher-managed option %s; use its dedicated setting\n' \ + "${argument%%=*}" >&2 + exit 2 + ;; + esac +done +# Preserve literal wildcard characters and never evaluate shell expressions. +lmcache_server+=("${server_extra_argv[@]}") + +lmcache_pid= +vllm_pid= + +signal_children() { + local signal=$1 + local pid + for pid in "${vllm_pid}" "${lmcache_pid}"; do + if [[ -n ${pid} ]] && kill -0 "${pid}" 2>/dev/null; then + kill -"${signal}" "${pid}" 2>/dev/null || true + fi + done +} + +wait_children() { + local pid + set +e + for pid in "${vllm_pid}" "${lmcache_pid}"; do + if [[ -n ${pid} ]]; then + wait "${pid}" 2>/dev/null + fi + done + set -e +} + +handle_signal() { + local status=$1 + trap - TERM INT + signal_children TERM + wait_children + exit "${status}" +} + +trap 'handle_signal 143' TERM +trap 'handle_signal 130' INT + +vllm_extra_args=() + +# max_num_scheduled_tokens limits target-model work, while +# max_num_batched_tokens includes any additional drafter query rows. Keeping +# the limits separate lets DFlash end target prefills on LMCache chunk +# boundaries without reducing the qualified 4096-token target budget. +draft_slots_per_request=0 +case "${SPECULATOR:-mtp}" in + dflash | dflash2) + num_speculative_tokens=${NUM_SPECULATIVE_TOKENS:-7} + if [[ ! ${num_speculative_tokens} =~ ^[0-9]+$ ]]; then + printf 'NUM_SPECULATIVE_TOKENS must be a non-negative integer; got %s\n' \ + "${num_speculative_tokens}" >&2 + exit 2 + fi + draft_slots_per_request=${num_speculative_tokens} + ;; + mtp) ;; + *) + printf 'SPECULATOR must be mtp, dflash, or dflash2; got %s\n' \ + "${SPECULATOR}" >&2 + exit 2 + ;; +esac +readonly input_token_budget=$(( + target_token_budget + draft_slots_per_request * max_num_seqs +)) +export MAX_NUM_BATCHED_TOKENS=${input_token_budget} +export KV_CACHE_DTYPE=${vllm_kv_cache_dtype} +export VLLM_GLM53_SPLIT_TARGET_BLOCK_SIZE="${VLLM_GLM53_SPLIT_TARGET_BLOCK_SIZE:-auto}" +export VLLM_GLM53_SPLIT_MAMBA_BLOCK_SIZE="${VLLM_GLM53_SPLIT_MAMBA_BLOCK_SIZE:-auto}" +vllm_extra_args+=(--max-num-scheduled-tokens "${target_token_budget}") + +if [[ ${transfer_mode} != engine_driven ]]; then + readonly interposer=/opt/lmcache/lib/liblmcache_cumem_shareable.so + if [[ ! -s ${interposer} ]]; then + printf 'LMCache cuMem transfer requires a nonempty interposer at %s\n' \ + "${interposer}" >&2 + exit 2 + fi + if ! nm -D "${interposer}" | awk '$3 == "cuMemCreate" { found=1 } END { exit !found }'; then + printf 'LMCache cuMem interposer does not export cuMemCreate: %s\n' \ + "${interposer}" >&2 + exit 2 + fi + if [[ -L ${broker_dir} ]]; then + printf 'LMCACHE_CUMEM_BROKER_DIR must not be a symlink: %s\n' \ + "${broker_dir}" >&2 + exit 2 + fi + mkdir -p "${broker_dir}" + chmod 700 "${broker_dir}" + if [[ ! -O ${broker_dir} || ! -w ${broker_dir} || ! -x ${broker_dir} ]]; then + printf 'LMCACHE_CUMEM_BROKER_DIR must be owned and writable by the serving user: %s\n' \ + "${broker_dir}" >&2 + exit 2 + fi + # Both the LMCache sidecar and vLLM workers must inherit the same path. The + # standard container recipe mounts /cache, so the default also remains valid + # if the two processes are later separated into distinct mount namespaces. + export LMCACHE_CUMEM_BROKER_DIR="${broker_dir}" + export LD_PRELOAD="${interposer}${LD_PRELOAD:+:${LD_PRELOAD}}" + vllm_extra_args+=(--enable-cumem-allocator) +fi + +# The CPU-only engine-driven sidecar has no CUDA context on serving GPUs. The +# 0.950 profile leaves measured long-prefill headroom while allocating more KV +# capacity than the direct-transfer profile; retain an explicit operator +# override when supplied. +if [[ ${transfer_mode} == engine_driven && -z ${GPU_MEMORY_UTILIZATION+x} ]]; then + export GPU_MEMORY_UTILIZATION=0.950 +fi + +lmcache_server_command=("${lmcache_server[@]}") +if [[ ${transfer_mode} == engine_driven ]]; then + # The standalone cache server manages CPU and filesystem storage only. GPU + # gather and scatter operations execute in the existing vLLM worker + # processes, so creating an additional CUDA context in the sidecar consumes + # device memory without providing a transfer service. + lmcache_server_env=${LMCACHE_SERVER_ENV-CUDA_VISIBLE_DEVICES= CUDA_MODULE_LOADING=LAZY} + read -r -a lmcache_server_env_args <<< "${lmcache_server_env}" + for assignment in "${lmcache_server_env_args[@]}"; do + if [[ ! ${assignment} =~ ^[A-Za-z_][A-Za-z0-9_]*=.*$ ]]; then + printf 'LMCACHE_SERVER_ENV contains an invalid assignment: %s\n' \ + "${assignment}" >&2 + exit 2 + fi + done + lmcache_server_command=( + env "${lmcache_server_env_args[@]}" "${lmcache_server[@]}" + ) +fi + +"${lmcache_server_command[@]}" & +lmcache_pid=$! + +readonly startup_start=${SECONDS} +until curl --fail --silent --show-error --max-time 2 "${health_url}" >/dev/null; do + if ! kill -0 "${lmcache_pid}" 2>/dev/null; then + set +e + wait "${lmcache_pid}" + status=$? + set -e + printf 'LMCache server exited before becoming ready with status %s\n' \ + "${status}" >&2 + if ((status == 0)); then + status=1 + fi + exit "${status}" + fi + if ((SECONDS - startup_start >= startup_timeout)); then + printf 'LMCache server did not become ready within %s seconds\n' \ + "${startup_timeout}" >&2 + signal_children TERM + wait_children + exit 1 + fi + sleep 0.25 +done + +connector_config=$(printf \ + '{"kv_connector":"LMCacheMPConnector","kv_connector_module_path":"lmcache.integration.vllm.lmcache_mp_connector","kv_role":"kv_both","kv_connector_extra_config":{"lmcache.mp.host":"%s","lmcache.mp.port":%s,"lmcache.mp.mp_transfer_mode":"%s"}}' \ + "${mp_host}" "${mp_port}" "${transfer_mode}") + +retention_interval=${chunk_size} +if [[ -n ${checkpoint_identity} ]]; then + connector_config=$(jq -c --argjson identity "${checkpoint_identity}" ' + .kv_connector = "LMCacheRecurrentCheckpointConnector" | + .kv_connector_module_path = "lmcache.integration.vllm.recurrent_checkpoint_connector" | + .kv_connector_extra_config["lmcache.mp.checkpoint_identity"] = $identity + ' <<< "${connector_config}") + if ((checkpoint_policy_supplied == 0)); then + vllm_extra_args+=(--recurrent-checkpoint-policy request_boundaries) + fi + retention_interval=0 +fi + +# Independent aligned objects require recurrent state at every object boundary. +# Atomic semantic bundles instead carry their own exact target/draft endpoint; +# they must not force additional periodic recurrent checkpoints. +vllm_extra_args+=( + --prefix-cache-retention-interval "${retention_interval}" + --kv-transfer-config "${connector_config}" +) + +"${base_launcher}" "$@" "${vllm_extra_args[@]}" & +vllm_pid=$! + +set +e +wait -n -p completed_pid "${lmcache_pid}" "${vllm_pid}" +status=$? +set -e +if [[ ${completed_pid:-} == "${lmcache_pid}" && ${status} == 0 ]]; then + printf 'LMCache server exited while vLLM was still running\n' >&2 + status=1 +fi +signal_children TERM +wait_children +exit "${status}" diff --git a/recipes/glm53/serve-glm53-flash-lmcache.sh b/recipes/glm53/serve-glm53-flash-lmcache.sh new file mode 100644 index 00000000..b69f097f --- /dev/null +++ b/recipes/glm53/serve-glm53-flash-lmcache.sh @@ -0,0 +1,358 @@ +#!/usr/bin/env bash +set -euo pipefail + +readonly base_launcher=/usr/local/bin/serve-glm53-flash-nvfp4-dflash2.sh + +# LMCache is opt-in so the default entrypoint preserves the qualified GLM-5.3 +# serving command, allocator, scheduler, and backend configuration exactly. +if [[ ${LMCACHE_ENABLED:-0} == 0 ]]; then + exec "${base_launcher}" "$@" +fi +if [[ ${LMCACHE_ENABLED} != 1 ]]; then + printf 'LMCACHE_ENABLED must be 0 or 1; got %s\n' "${LMCACHE_ENABLED}" >&2 + exit 2 +fi + +require_positive_integer() { + local name=$1 + local value=$2 + if [[ ! ${value} =~ ^[1-9][0-9]*$ ]]; then + printf '%s must be a positive integer; got %s\n' "${name}" "${value}" >&2 + exit 2 + fi +} + +readonly mp_host=${LMCACHE_MP_HOST:-127.0.0.1} +readonly http_host=${LMCACHE_HTTP_HOST:-127.0.0.1} +readonly mp_port=${LMCACHE_MP_PORT:-5555} +readonly http_port=${LMCACHE_HTTP_PORT:-8085} +readonly prometheus_port=${LMCACHE_PROMETHEUS_PORT:-9095} +readonly startup_timeout=${LMCACHE_STARTUP_TIMEOUT_SECONDS:-120} +readonly transfer_mode=${LMCACHE_TRANSFER_MODE:-lmcache_driven} +readonly instance_id=${LMCACHE_INSTANCE_ID:-glm53-jovian-judgement-lmcache} +readonly broker_dir=${LMCACHE_CUMEM_BROKER_DIR:-/cache/lmcache-cumem} +readonly prefetch_policy=${LMCACHE_L2_PREFETCH_POLICY:-retain} +readonly server_extra_args=${LMCACHE_SERVER_EXTRA_ARGS:-} +readonly chunk_size=${LMCACHE_CHUNK_SIZE:-4096} +readonly target_token_budget=${LMCACHE_TARGET_TOKEN_BUDGET:-${MAX_NUM_BATCHED_TOKENS:-4096}} +readonly max_num_seqs=${MAX_NUM_SEQS:-16} +readonly min_shm_gib=${LMCACHE_MIN_SHM_GIB:-96} +readonly lmcache_kv_cache_dtype=${LMCACHE_KV_CACHE_DTYPE:-nvfp4_ds_mla} + +require_positive_integer LMCACHE_MP_PORT "${mp_port}" +require_positive_integer LMCACHE_HTTP_PORT "${http_port}" +require_positive_integer LMCACHE_PROMETHEUS_PORT "${prometheus_port}" +require_positive_integer LMCACHE_STARTUP_TIMEOUT_SECONDS "${startup_timeout}" +require_positive_integer LMCACHE_CHUNK_SIZE "${chunk_size}" +require_positive_integer LMCACHE_TARGET_TOKEN_BUDGET "${target_token_budget}" +require_positive_integer MAX_NUM_SEQS "${max_num_seqs}" +require_positive_integer LMCACHE_MIN_SHM_GIB "${min_shm_gib}" +if [[ ! ${mp_host} =~ ^[A-Za-z0-9_.:-]+$ ]]; then + printf 'LMCACHE_MP_HOST contains unsupported characters: %s\n' \ + "${mp_host}" >&2 + exit 2 +fi +if [[ ! ${http_host} =~ ^[A-Za-z0-9_.:][A-Za-z0-9_.:-]*$ ]]; then + printf 'LMCACHE_HTTP_HOST contains unsupported characters: %s\n' \ + "${http_host}" >&2 + exit 2 +fi +# A wildcard bind needs a concrete loopback address for the readiness probe. +http_probe_host=${http_host} +case "${http_host}" in + 0.0.0.0) http_probe_host=127.0.0.1 ;; + ::) http_probe_host=::1 ;; +esac +if [[ ${http_probe_host} == *:* ]]; then + http_probe_host="[${http_probe_host}]" +fi +readonly health_url="http://${http_probe_host}:${http_port}/healthcheck" +http_loopback=0 +if [[ ${http_host} == ::1 || ${http_host} == localhost ]]; then + http_loopback=1 +elif [[ ${http_host} =~ ^127\.([0-9]{1,3})\.([0-9]{1,3})\.([0-9]{1,3})$ ]] && + ((10#${BASH_REMATCH[1]} <= 255 && 10#${BASH_REMATCH[2]} <= 255 && + 10#${BASH_REMATCH[3]} <= 255)); then + http_loopback=1 +fi +if ((http_loopback == 0)); then + printf '%s\n' 'LMCache HTTP includes administrative APIs; restrict access to a trusted network or authenticated proxy.' >&2 +fi + +case "${transfer_mode}" in + lmcache_driven | auto | engine_driven) ;; + *) + printf 'LMCACHE_TRANSFER_MODE must be lmcache_driven, auto, or engine_driven; got %s\n' \ + "${transfer_mode}" >&2 + exit 2 + ;; +esac + +shm_name=${LMCACHE_SHM_NAME:-} +if [[ ${transfer_mode} == engine_driven && -z ${shm_name} ]]; then + # Concurrent host-network sidecars have distinct MP ports. Include that port + # in the arena name so their worker-owned copies cannot alias another pool. + safe_instance_id=${instance_id//[^A-Za-z0-9._-]/_} + shm_name="lmcache-${safe_instance_id}-${mp_port}" +fi +readonly shm_name +if [[ -n ${shm_name} && ! ${shm_name} =~ ^[A-Za-z0-9._-]+$ ]]; then + printf 'LMCACHE_SHM_NAME contains unsupported characters: %s\n' \ + "${shm_name}" >&2 + exit 2 +fi + +case "${prefetch_policy}" in + default | retain) ;; + *) + printf 'LMCACHE_L2_PREFETCH_POLICY must be default or retain; got %s\n' \ + "${prefetch_policy}" >&2 + exit 2 + ;; +esac + +case "${lmcache_kv_cache_dtype}" in + fp8 | fp8_e4m3 | fp8_ds_mla | nvfp4_ds_mla) ;; + *) + printf 'LMCACHE_KV_CACHE_DTYPE must be fp8, fp8_e4m3, fp8_ds_mla, or nvfp4_ds_mla; got %s\n' \ + "${lmcache_kv_cache_dtype}" >&2 + exit 2 + ;; +esac + +shm_bytes=$(df -B1 --output=size /dev/shm | awk 'NR == 2 {print $1}') +readonly shm_bytes +readonly min_shm_bytes=$((min_shm_gib * 1024 * 1024 * 1024)) +if [[ ! ${shm_bytes} =~ ^[0-9]+$ ]] || ((shm_bytes < min_shm_bytes)); then + printf 'LMCache native transfer requires a private /dev/shm of at least %s GiB; got %s bytes\n' \ + "${min_shm_gib}" "${shm_bytes:-unknown}" >&2 + exit 2 +fi + +lmcache_server=( + /opt/venv/bin/lmcache server + --instance-id "${instance_id}" + --host "${mp_host}" + --port "${mp_port}" + --chunk-size "${chunk_size}" + --max-workers "${LMCACHE_MAX_WORKERS:-8}" + --max-gpu-workers "${LMCACHE_MAX_GPU_WORKERS:-8}" + --max-cpu-workers "${LMCACHE_MAX_CPU_WORKERS:-16}" + --hash-algorithm blake3 + --supported-transfer-mode "${transfer_mode}" + --separate-object-groups + --l1-size-gb "${LMCACHE_L1_SIZE_GB:-64}" + --l1-init-size-gb "${LMCACHE_L1_INIT_SIZE_GB:-2}" + --eviction-policy LRU + --l2-prefetch-policy "${prefetch_policy}" + --http-host "${http_host}" + --http-port "${http_port}" + --prometheus-port "${prometheus_port}" +) + +if [[ ${transfer_mode} == engine_driven ]]; then + # Worker-owned SHM copies require stable views into a preallocated L1 arena. + lmcache_server+=(--no-l1-use-lazy --shm-name "${shm_name}") +else + lmcache_server+=(--l1-use-lazy) +fi + +case "${LMCACHE_L2_ENABLED:-1}" in + 0) ;; + 1) + l2_path=${LMCACHE_L2_PATH:-/lmcache-l2} + if [[ ${l2_path} != /* || ${l2_path} == *['"\']* ]]; then + printf 'LMCACHE_L2_PATH must be an absolute path without quotes or backslashes\n' >&2 + exit 2 + fi + mkdir -p "${l2_path}" + l2_config=${LMCACHE_L2_CONFIG:-} + if [[ -z ${l2_config} ]]; then + printf -v l2_config \ + '{"type":"fs_native","base_path":"%s","num_workers":%s,"use_odirect":false,"max_capacity_gb":%s,"eviction":{"eviction_policy":"LRU","trigger_watermark":0.8,"eviction_ratio":0.2}}' \ + "${l2_path}" "${LMCACHE_L2_WORKERS:-8}" \ + "${LMCACHE_L2_MAX_CAPACITY_GB:-512}" + fi + /opt/venv/bin/python -c \ + 'import json, sys; value=json.loads(sys.argv[1]); assert isinstance(value, dict)' \ + "${l2_config}" + lmcache_server+=(--l2-adapter "${l2_config}") + ;; + *) + printf 'LMCACHE_L2_ENABLED must be 0 or 1; got %s\n' \ + "${LMCACHE_L2_ENABLED}" >&2 + exit 2 + ;; +esac + +if [[ ${server_extra_args} == *$'\n'* || ${server_extra_args} == *$'\r'* ]]; then + printf 'LMCACHE_SERVER_EXTRA_ARGS must be one whitespace-separated line\n' >&2 + exit 2 +fi +if [[ ${prefetch_policy} == retain && ${LMCACHE_L2_ENABLED:-1} == 1 ]]; then + lmcache_server+=(--emergency-evict-for-prefetch) +fi +server_extra_argv=() +read -r -a server_extra_argv <<< "${server_extra_args}" +for argument in "${server_extra_argv[@]}"; do + case "${argument%%=*}" in + --instance-id | --host | --port | --http-host | --http-port | --prometheus-port | \ + --supported-transfer-mode | --chunk-size | --hash-algorithm | \ + --separate-object-groups | --no-separate-object-groups | --shm-name | \ + --l1-use-lazy | --no-l1-use-lazy | --checkpoint-index-path | --l2-prefetch-policy | \ + --emergency-evict-for-prefetch | --no-emergency-evict-for-prefetch) + printf 'LMCACHE_SERVER_EXTRA_ARGS cannot override launcher-managed option %s; use its dedicated setting\n' \ + "${argument%%=*}" >&2 + exit 2 + ;; + esac +done +# Preserve literal wildcard characters and never evaluate shell expressions. +lmcache_server+=("${server_extra_argv[@]}") + +lmcache_pid= +vllm_pid= + +signal_children() { + local signal=$1 + local pid + for pid in "${vllm_pid}" "${lmcache_pid}"; do + if [[ -n ${pid} ]] && kill -0 "${pid}" 2>/dev/null; then + kill -"${signal}" "${pid}" 2>/dev/null || true + fi + done +} + +wait_children() { + local pid + set +e + for pid in "${vllm_pid}" "${lmcache_pid}"; do + if [[ -n ${pid} ]]; then + wait "${pid}" 2>/dev/null + fi + done + set -e +} + +handle_signal() { + local status=$1 + trap - TERM INT + signal_children TERM + wait_children + exit "${status}" +} + +trap 'handle_signal 143' TERM +trap 'handle_signal 130' INT + +vllm_extra_args=() + +# max_num_scheduled_tokens limits target-model work, while +# max_num_batched_tokens includes any additional drafter query rows. Keeping +# the limits separate lets DFlash end target prefills on LMCache chunk +# boundaries without reducing the qualified 4096-token target budget. +draft_slots_per_request=0 +case "${SPECULATOR:-mtp}" in + dflash | dflash2) + num_speculative_tokens=${NUM_SPECULATIVE_TOKENS:-7} + if [[ ! ${num_speculative_tokens} =~ ^[0-9]+$ ]]; then + printf 'NUM_SPECULATIVE_TOKENS must be a non-negative integer; got %s\n' \ + "${num_speculative_tokens}" >&2 + exit 2 + fi + draft_slots_per_request=${num_speculative_tokens} + ;; + mtp) ;; + *) + printf 'SPECULATOR must be mtp, dflash, or dflash2; got %s\n' \ + "${SPECULATOR}" >&2 + exit 2 + ;; +esac +readonly input_token_budget=$(( + target_token_budget + draft_slots_per_request * max_num_seqs +)) +export MAX_NUM_BATCHED_TOKENS=${input_token_budget} +case "${lmcache_kv_cache_dtype}" in + fp8_ds_mla) export KV_CACHE_DTYPE=fp8 ;; + *) export KV_CACHE_DTYPE=${lmcache_kv_cache_dtype} ;; +esac +export VLLM_GLM53_SPLIT_TARGET_BLOCK_SIZE="${VLLM_GLM53_SPLIT_TARGET_BLOCK_SIZE:-auto}" +export VLLM_GLM53_SPLIT_MAMBA_BLOCK_SIZE="${VLLM_GLM53_SPLIT_MAMBA_BLOCK_SIZE:-auto}" +vllm_extra_args+=(--max-num-scheduled-tokens "${target_token_budget}") + +if [[ ${transfer_mode} != engine_driven ]]; then + if [[ -L ${broker_dir} ]]; then + printf 'LMCACHE_CUMEM_BROKER_DIR must not be a symlink: %s\n' \ + "${broker_dir}" >&2 + exit 2 + fi + mkdir -p "${broker_dir}" + chmod 700 "${broker_dir}" + if [[ ! -O ${broker_dir} || ! -w ${broker_dir} || ! -x ${broker_dir} ]]; then + printf 'LMCACHE_CUMEM_BROKER_DIR must be owned and writable by the serving user: %s\n' \ + "${broker_dir}" >&2 + exit 2 + fi + # Both the LMCache sidecar and vLLM workers must inherit the same path. The + # standard container recipe mounts /cache, so the default also remains valid + # if the two processes are later separated into distinct mount namespaces. + export LMCACHE_CUMEM_BROKER_DIR="${broker_dir}" + export LD_PRELOAD="/opt/lmcache/lib/liblmcache_cumem_shareable.so${LD_PRELOAD:+:${LD_PRELOAD}}" + vllm_extra_args+=(--enable-cumem-allocator) +fi + +"${lmcache_server[@]}" & +lmcache_pid=$! + +readonly startup_start=${SECONDS} +until curl --fail --silent --show-error --max-time 2 "${health_url}" >/dev/null; do + if ! kill -0 "${lmcache_pid}" 2>/dev/null; then + set +e + wait "${lmcache_pid}" + status=$? + set -e + printf 'LMCache server exited before becoming ready with status %s\n' \ + "${status}" >&2 + if ((status == 0)); then + status=1 + fi + exit "${status}" + fi + if ((SECONDS - startup_start >= startup_timeout)); then + printf 'LMCache server did not become ready within %s seconds\n' \ + "${startup_timeout}" >&2 + signal_children TERM + wait_children + exit 1 + fi + sleep 0.25 +done + +connector_config=$(printf \ + '{"kv_connector":"LMCacheMPConnector","kv_connector_module_path":"lmcache.integration.vllm.lmcache_mp_connector","kv_role":"kv_both","kv_connector_extra_config":{"lmcache.mp.host":"%s","lmcache.mp.port":%s,"lmcache.mp.mp_transfer_mode":"%s"}}' \ + "${mp_host}" "${mp_port}" "${transfer_mode}") + +# A hybrid recurrent cache object is usable only when vLLM retains the exact +# recurrent state at the same token boundary as the LMCache object. The vLLM +# argument also validates that the object size is compatible with the resolved +# scheduler block size. +vllm_extra_args+=( + --prefix-cache-retention-interval "${chunk_size}" + --kv-transfer-config "${connector_config}" +) + +"${base_launcher}" "$@" "${vllm_extra_args[@]}" & +vllm_pid=$! + +set +e +wait -n -p completed_pid "${lmcache_pid}" "${vllm_pid}" +status=$? +set -e +if [[ ${completed_pid:-} == "${lmcache_pid}" && ${status} == 0 ]]; then + printf 'LMCache server exited while vLLM was still running\n' >&2 + status=1 +fi +signal_children TERM +wait_children +exit "${status}" diff --git a/recipes/glm53/serve-glm53-flash-nvfp4-dflash2-scheduler-qos.sh b/recipes/glm53/serve-glm53-flash-nvfp4-dflash2-scheduler-qos.sh new file mode 100644 index 00000000..a3f9d027 --- /dev/null +++ b/recipes/glm53/serve-glm53-flash-nvfp4-dflash2-scheduler-qos.sh @@ -0,0 +1,217 @@ +#!/usr/bin/env bash +set -euo pipefail + +# DFlash K7 verifies eight target rows per request, while MTP3 verifies four. +# These capture sizes keep C1/C8/C12/C24 target batches on exact full CUDA +# graphs. Other request counts use vLLM's next captured padded size. +readonly base_launcher=/usr/local/libexec/serve-glm53-flash-nvfp4-dflash2.sh +capture_sizes=${CUDAGRAPH_CAPTURE_SIZES:-'1 2 4 8 16 32 40 48 64 96 128 192 256'} + +export MAX_CUDAGRAPH_CAPTURE_SIZE=${MAX_CUDAGRAPH_CAPTURE_SIZE:-256} + +fail() { + printf '%s\n' "$1" >&2 + exit 2 +} + +require_nonnegative_integer() { + local name=$1 + local value=$2 + [[ ${value} =~ ^[0-9]+$ ]] || + fail "${name} must be a non-negative integer; got ${value}" +} + +require_open_unit_interval() { + local name=$1 + local value=$2 + require_positive_finite "${name}" "${value}" + awk -v value="${value}" 'BEGIN { exit !(value < 1.0) }' || + fail "${name} must be greater than zero and less than one; got ${value}" +} + +require_positive_finite() { + local name=$1 value=$2 + [[ ${value} =~ ^[+]?[0-9]*\.?[0-9]+([eE][+-]?[0-9]+)?$ || + ${value} =~ ^[+]?[0-9]+\.([eE][+-]?[0-9]+)?$ ]] && + awk -v value="${value}" 'BEGIN { + exit !(value > 0 && tolower(sprintf("%.17g", value)) !~ /inf|nan/) + }' || fail "${name} must be positive and finite; got ${value}" +} + +has_cli_option() { + local expected=$1 + shift + local argument + for argument in "$@"; do + if [[ ${argument} == "${expected}" || ${argument} == "${expected}="* ]]; then + return 0 + fi + done + return 1 +} + +runtime_args=() +launcher_args=("$@") + +if (($# == 1)) && [[ $1 == --help || $1 == -h ]]; then + printf '%s\n' 'GLM-5.3-Flash scheduler controls (environment -> vLLM CLI): + PREFILL_COMPUTE_SHARE -> --prefill-compute-share: auto or 0 < float < 1 + PREFILL_COMPUTE_HALF_LIFE -> --prefill-compute-half-life: smooth, responsive, or positive finite seconds; requires auto share + MAX_PARALLEL_PREFILLS -> --max-parallel-prefills: auto or positive integer + PREFILL_POLICY -> --prefill-policy: round-robin or decode-aware + DECODE_REFILL_TARGET -> --decode-refill-target: auto or positive integer + +Explicit CLI values override environment values. Unset controls retain vLLM defaults. +The community image sets fixed share 0.4, interval 1, and retains one prefill lane. +Compute-share fairness requires --prefill-schedule-interval 1. +FAIRNESS_ENGINE=none disables inherited share; an explicit CLI share still applies. +FAIRNESS_ENGINE=micro_slicing is unsupported. +Use DRY_RUN=1 with CACHE_MODE=vram to inspect the complete serving command.' + exit 0 +fi + +# Resolve the effective value before validation. Invalid shadowed environment +# values must not reject a valid command-line override. +resolve_option() { + local option=$1 environment_value=$2 emit_environment=${3:-1} + local index argument seen=0 + resolved_value=${environment_value} + for ((index = 0; index < ${#launcher_args[@]}; index++)); do + argument=${launcher_args[index]} + if [[ ${argument} == "${option}" || ${argument} == "${option}="* ]]; then + ((seen == 0)) || fail "Specify ${option} only once" + seen=1 + if [[ ${argument} == "${option}" ]]; then + ((index + 1 < ${#launcher_args[@]})) || fail "${option} requires a value" + resolved_value=${launcher_args[index + 1]} + else + resolved_value=${argument#*=} + fi + [[ -n ${resolved_value} && ${resolved_value} != --* ]] || + fail "${option} requires a value" + fi + done + if ((seen == 0 && emit_environment)) && [[ -n ${environment_value} ]]; then + runtime_args+=("${option}" "${environment_value}") + fi +} + +# Compute-share fairness charges measured model-execution time to prefill and +# decode. ``PREFILL_COMPUTE_SHARE`` is the vLLM scheduler interface. The +# ``FAIRNESS_ENGINE`` environment variable remains a launcher compatibility +# control for deployments created before the scheduler exposed compute share +# directly. +fairness_engine=${FAIRNESS_ENGINE:-} +share_environment=${PREFILL_COMPUTE_SHARE:-} +case "${fairness_engine}" in + '' | compute_share) ;; + none) share_environment= ;; + micro_slicing) + fail 'FAIRNESS_ENGINE=micro_slicing is unsupported by this vLLM scheduler; use compute_share or none' + ;; + *) + fail "FAIRNESS_ENGINE must be none, compute_share, or micro_slicing; got ${fairness_engine}" + ;; +esac + +resolve_option --prefill-compute-share "${share_environment}" +prefill_compute_share=${resolved_value} +if [[ ${fairness_engine} == compute_share && -z ${prefill_compute_share} ]]; then + fail 'PREFILL_COMPUTE_SHARE or --prefill-compute-share is required for FAIRNESS_ENGINE=compute_share' +fi +if [[ -n ${prefill_compute_share} && ${prefill_compute_share} != auto ]]; then + require_open_unit_interval PREFILL_COMPUTE_SHARE "${prefill_compute_share}" +fi +resolve_option --prefill-schedule-interval "${PREFILL_SCHEDULE_INTERVAL:-8}" 0 +if [[ -n ${prefill_compute_share} && ${resolved_value} != 1 ]]; then + fail 'PREFILL_SCHEDULE_INTERVAL/--prefill-schedule-interval must be 1 when compute-share fairness is enabled' +fi + +resolve_option --prefill-compute-half-life "${PREFILL_COMPUTE_HALF_LIFE:-}" +if [[ -n ${resolved_value} ]]; then + [[ ${prefill_compute_share} == auto ]] || + fail 'PREFILL_COMPUTE_HALF_LIFE requires effective PREFILL_COMPUTE_SHARE=auto' + case "${resolved_value}" in + smooth | responsive) ;; + *) require_positive_finite PREFILL_COMPUTE_HALF_LIFE "${resolved_value}" ;; + esac +fi + +resolve_option --max-parallel-prefills "${MAX_PARALLEL_PREFILLS:-}" +if [[ -n ${resolved_value} && ${resolved_value} != auto ]]; then + [[ ${resolved_value} =~ ^[1-9][0-9]*$ ]] || + fail "MAX_PARALLEL_PREFILLS must be auto or a positive integer; got ${resolved_value}" +fi +resolve_option --prefill-policy "${PREFILL_POLICY:-}" +case "${resolved_value}" in + '' | round-robin | decode-aware) ;; + *) fail "PREFILL_POLICY must be round-robin or decode-aware; got ${resolved_value}" ;; +esac +resolve_option --decode-refill-target "${DECODE_REFILL_TARGET:-}" +if [[ -n ${resolved_value} && ${resolved_value} != auto ]]; then + [[ ${resolved_value} =~ ^[1-9][0-9]*$ ]] || + fail "DECODE_REFILL_TARGET must be auto or a positive integer; got ${resolved_value}" +fi + +# A caller-supplied CLI value is authoritative and must not be duplicated. +if [[ ${capture_sizes} != none ]] && + ! has_cli_option --cudagraph-capture-sizes "$@"; then + read -r -a capture_size_args <<<"${capture_sizes}" + if ((${#capture_size_args[@]} == 0)); then + fail 'CUDAGRAPH_CAPTURE_SIZES must contain positive integers or "none"' + fi + + previous=0 + for size in "${capture_size_args[@]}"; do + if [[ ! ${size} =~ ^[1-9][0-9]*$ ]]; then + fail "Invalid CUDA graph capture size: ${size}" + fi + if ((size <= previous)); then + fail "CUDAGRAPH_CAPTURE_SIZES must be strictly increasing; got ${size} after ${previous}" + fi + if ((size > MAX_CUDAGRAPH_CAPTURE_SIZE)); then + fail "CUDA graph capture size ${size} exceeds MAX_CUDAGRAPH_CAPTURE_SIZE=${MAX_CUDAGRAPH_CAPTURE_SIZE}" + fi + previous=${size} + done + runtime_args+=(--cudagraph-capture-sizes "${capture_size_args[@]}") +fi + +# Mixed decode/prefill controls are disabled by default. Nonzero values opt in +# to bounded prefill quanta, partial-prefill admission, and decode-burst QoS. +max_num_prefill_tokens_per_step=${MAX_NUM_PREFILL_TOKENS_PER_STEP:-0} +max_num_partial_prefills=${MAX_NUM_PARTIAL_PREFILLS:-0} +decode_prefill_min_decode_steps=${DECODE_PREFILL_MIN_DECODE_STEPS:-0} +decode_prefill_max_wait_ms=${DECODE_PREFILL_MAX_WAIT_MS:-0} + +require_nonnegative_integer MAX_NUM_PREFILL_TOKENS_PER_STEP \ + "${max_num_prefill_tokens_per_step}" +require_nonnegative_integer MAX_NUM_PARTIAL_PREFILLS \ + "${max_num_partial_prefills}" +require_nonnegative_integer DECODE_PREFILL_MIN_DECODE_STEPS \ + "${decode_prefill_min_decode_steps}" +require_nonnegative_integer DECODE_PREFILL_MAX_WAIT_MS \ + "${decode_prefill_max_wait_ms}" + +if ((max_num_prefill_tokens_per_step > 0)) && + ! has_cli_option --max-num-prefill-tokens-per-step "$@"; then + runtime_args+=( + --max-num-prefill-tokens-per-step "${max_num_prefill_tokens_per_step}" + ) +fi +if ((max_num_partial_prefills > 0)) && + ! has_cli_option --max-num-partial-prefills "$@"; then + runtime_args+=(--max-num-partial-prefills "${max_num_partial_prefills}") +fi +if ((decode_prefill_min_decode_steps > 0)) && + ! has_cli_option --decode-prefill-min-decode-steps "$@"; then + runtime_args+=( + --decode-prefill-min-decode-steps "${decode_prefill_min_decode_steps}" + ) +fi +if ((decode_prefill_max_wait_ms > 0)) && + ! has_cli_option --decode-prefill-max-wait-ms "$@"; then + runtime_args+=(--decode-prefill-max-wait-ms "${decode_prefill_max_wait_ms}") +fi + +exec "${base_launcher}" "$@" "${runtime_args[@]}" diff --git a/recipes/glm53/serve-glm53-flash-nvfp4-dflash2.sh b/recipes/glm53/serve-glm53-flash-nvfp4-dflash2.sh new file mode 100644 index 00000000..2d8432a2 --- /dev/null +++ b/recipes/glm53/serve-glm53-flash-nvfp4-dflash2.sh @@ -0,0 +1,328 @@ +#!/usr/bin/env bash +set -euo pipefail + +# This single-node TP launcher does not use an external NCCL topology file. +# NCCL interprets an empty graph-file value as a path, so remove it completely. +unset NCCL_GRAPH_FILE + +# Serving is GPU-bound for the qualified TP4 paths. One OpenMP thread avoids +# spin-wait contention; measurements with 2, 4, and 8 threads do not improve +# target-step or output throughput. Operators may still override the value. +export OMP_NUM_THREADS="${OMP_NUM_THREADS:-1}" + +model=${MODEL:-local-inference-lab/GLM-5.3-Flash-NVFP4} +if (($# > 0)) && [[ "$1" != -* ]]; then + model=$1 + shift +fi +model_revision=${MODEL_REVISION-} + +served_model_name=${SERVED_MODEL_NAME:-GLM-5.3-Flash-NVFP4} +host=${HOST:-0.0.0.0} +port=${PORT:-8000} +tp=${TP:-4} +dcp=${DCP:-1} +cp_kv_cache_interleave_size=${CP_KV_CACHE_INTERLEAVE_SIZE:-4} +max_num_seqs=${MAX_NUM_SEQS:-32} +max_model_len=${MAX_MODEL_LEN:-1048576} +max_num_batched_tokens=${MAX_NUM_BATCHED_TOKENS:-4096} +prefill_schedule_interval=${PREFILL_SCHEDULE_INTERVAL:-8} +prefill_interval_from_cli=0 +chat_defaults_from_cli=0 +generation_defaults_from_cli=0 +for argument in "$@"; do + case "${argument}" in + --prefill-schedule-interval | --prefill-schedule-interval=*) + prefill_interval_from_cli=1 + ;; + --default-chat-template-kwargs | --default-chat-template-kwargs=*) + chat_defaults_from_cli=1 + ;; + esac + case "${argument//_/-}" in + --generation-config | --generation-config=* | \ + --override-generation-config | --override-generation-config=* | \ + --override-generation-config.* | --config | --config=*) + generation_defaults_from_cli=1 + ;; + esac +done +max_cudagraph_capture_size=${MAX_CUDAGRAPH_CAPTURE_SIZE:-128} +gpu_memory_utilization=${GPU_MEMORY_UTILIZATION:-0.93} +load_format=${LOAD_FORMAT:-instanttensor} +speculator=${SPECULATOR:-mtp} +dflash_model=${DFLASH_MODEL:-local-inference-lab/GLM-5.3-Flash-DFlash2} +dflash_model_revision=${DFLASH_MODEL_REVISION-} +dflash_kv_cache_dtype=${DFLASH_KV_CACHE_DTYPE:-auto} +dflash_attention_backend=${DFLASH_ATTENTION_BACKEND:-FLASH_ATTN} +kv_cache_dtype=${KV_CACHE_DTYPE:-fp8} +attention_backend=${ATTENTION_BACKEND:-B12X} +moe_backend=${MOE_BACKEND:-b12x} +linear_backend=${LINEAR_BACKEND:-b12x} +mtp_attention_backend=${MTP_ATTENTION_BACKEND:-B12X} +mtp_moe_backend=${MTP_MOE_BACKEND:-marlin} +b12x_pcie_allreduce=${B12X_PCIE_ALLREDUCE:-1} +kda_decode_backend=${GLM53_KDA_DECODE_BACKEND:-auto} +kda_prefill_backend=${GLM53_KDA_PREFILL_BACKEND:-flashkda} +cudagraph_mode=${CUDAGRAPH_MODE:-FULL_AND_PIECEWISE} +dcp_ckv_gather=${DCP_CKV_GATHER:-auto} + +if [[ ! "${tp}" =~ ^[1-9][0-9]*$ || ! "${dcp}" =~ ^[1-9][0-9]*$ ]]; then + printf 'TP and DCP must be positive integers; got TP=%s DCP=%s\n' \ + "${tp}" "${dcp}" >&2 + exit 2 +fi +if ((tp % dcp != 0)); then + printf 'DCP must divide TP; got TP=%s DCP=%s\n' "${tp}" "${dcp}" >&2 + exit 2 +fi +if [[ ! "${cp_kv_cache_interleave_size}" =~ ^[1-9][0-9]*$ ]]; then + printf 'CP_KV_CACHE_INTERLEAVE_SIZE must be a positive integer; got %s\n' \ + "${cp_kv_cache_interleave_size}" >&2 + exit 2 +fi +if ((prefill_interval_from_cli == 0)) && + [[ ! "${prefill_schedule_interval}" =~ ^[1-9][0-9]*$ ]]; then + printf 'PREFILL_SCHEDULE_INTERVAL must be a positive integer; got %s\n' \ + "${prefill_schedule_interval}" >&2 + exit 2 +fi + +case "${dcp_ckv_gather}" in + auto) + # Independent target and recurrent pages let MTP and DFlash2 amortize + # full-CKV gathering at the qualified 4096-token scheduler budget. + if ((dcp > 1)); then + dcp_ckv_gather=1 + else + dcp_ckv_gather=0 + fi + ;; + 0 | 1) ;; + *) + printf 'DCP_CKV_GATHER must be auto, 0, or 1; got %s\n' \ + "${dcp_ckv_gather}" >&2 + exit 2 + ;; +esac + +# B12X DMA remains beneficial for DCP1. DCP full-CKV prefill uses PyNCCL for +# messages above the one-shot/two-shot ranges because the measured B12X DMA +# crossover is slower for that workload. An explicit operator value remains +# authoritative. +pcie_dma_min_bytes=${VLLM_PCIE_DMA_MIN_BYTES:-} +if [[ -z ${pcie_dma_min_bytes} ]]; then + if ((dcp > 1)); then + pcie_dma_min_bytes=off + else + pcie_dma_min_bytes=6MB + fi +fi + +case "${speculator}" in + mtp) + num_speculative_tokens=${NUM_SPECULATIVE_TOKENS:-${MTP:-0}} + ;; + dflash | dflash2) + # DFlash2 is trained for an eight-token block: one verified token and + # seven draft tokens. NUM_SPECULATIVE_TOKENS can qualify another width. + num_speculative_tokens=${NUM_SPECULATIVE_TOKENS:-7} + ;; + *) + printf 'SPECULATOR must be mtp, dflash, or dflash2; got %s\n' \ + "${speculator}" >&2 + exit 2 + ;; +esac + +if [[ ! "${num_speculative_tokens}" =~ ^[0-9]+$ ]]; then + printf 'NUM_SPECULATIVE_TOKENS/MTP must be a non-negative integer; got %s\n' \ + "${num_speculative_tokens}" >&2 + exit 2 +fi + +# A single-token target step is faster with quantized NVFP4 activations on the +# qualified RTX PRO 6000 path. Multi-token speculative verifiers retain B12X's +# profile-driven selection, which is faster for their four- and eight-row +# GEMMs. Explicit precision settings remain authoritative. +if [[ ${speculator} == mtp && ${num_speculative_tokens} == 0 && + -z ${VLLM_B12X_DENSE_ACTIVATION_MODE+x} && + -z ${VLLM_B12X_NVFP4_ACTIVATION_MODE+x} ]]; then + export VLLM_B12X_NVFP4_ACTIVATION_MODE=quantized +fi + +case "${b12x_pcie_allreduce}" in + 0 | 1) ;; + *) + printf 'B12X_PCIE_ALLREDUCE must be 0 or 1; got %s\n' \ + "${b12x_pcie_allreduce}" >&2 + exit 2 + ;; +esac + +case "${kda_decode_backend}" in + auto | b12x | triton) ;; + *) + printf 'GLM53_KDA_DECODE_BACKEND must be auto, b12x, or triton; got %s\n' \ + "${kda_decode_backend}" >&2 + exit 2 + ;; +esac + +case "${kda_prefill_backend}" in + auto | b12x | flashkda | triton) ;; + *) + printf 'GLM53_KDA_PREFILL_BACKEND must be auto, b12x, flashkda, or triton; got %s\n' \ + "${kda_prefill_backend}" >&2 + exit 2 + ;; +esac + +case "${dflash_attention_backend}" in + FLASHINFER | FLASH_ATTN) ;; + *) + printf 'DFLASH_ATTENTION_BACKEND must be FLASHINFER or FLASH_ATTN; got %s\n' \ + "${dflash_attention_backend}" >&2 + exit 2 + ;; +esac + +case "${kv_cache_dtype}" in + fp8 | fp8_e4m3 | fp8_ds_mla | nvfp4_ds_mla) ;; + *) + printf 'KV_CACHE_DTYPE must be fp8, fp8_e4m3, fp8_ds_mla, or nvfp4_ds_mla; got %s\n' \ + "${kv_cache_dtype}" >&2 + exit 2 + ;; +esac + +# Zero retains the target checkpoint's NVFP4 W4A4 routed-expert path. +export VLLM_B12X_MOE_FP4_FORCE_A16="${VLLM_B12X_MOE_FP4_FORCE_A16:-0}" +export VLLM_ENABLE_PCIE_ALLREDUCE="${b12x_pcie_allreduce}" +export VLLM_PCIE_ALLREDUCE_BACKEND=b12x +export VLLM_PCIE_DMA_MIN_BYTES="${pcie_dma_min_bytes}" +export VLLM_B12X_MLA_CKV_GATHER="${dcp_ckv_gather}" +# GLM-5.3 stores sparse target KV and recurrent state in separate allocations. +# The 2,048-token target page gives the qualified GPU-cache capacity without a +# measurable prefill or decode penalty. The recurrent page follows the target +# unless an external-cache launcher derives both from its transfer interval. +export VLLM_GLM53_SPLIT_TARGET_BLOCK_SIZE="${VLLM_GLM53_SPLIT_TARGET_BLOCK_SIZE:-2048}" +export VLLM_GLM53_SPLIT_MAMBA_BLOCK_SIZE="${VLLM_GLM53_SPLIT_MAMBA_BLOCK_SIZE:-auto}" +export VLLM_PLUGINS= + +revision_args=() +if [[ -n "${model_revision}" && "${model}" != /* ]]; then + revision_args=(--revision "${model_revision}") +fi + +cmd=( + /opt/venv/bin/vllm serve "${model}" + "${revision_args[@]}" + --served-model-name "${served_model_name}" + --host "${host}" + --port "${port}" + --tensor-parallel-size "${tp}" + --pipeline-parallel-size 1 + --decode-context-parallel-size "${dcp}" + # Glm5Next B12X C4 consumes cp_kv_cache_interleave_size, while generic + # FlashInfer DCP consumes dcp_kv_cache_interleave_size. Equal values preserve + # one rank-local cache layout for target and draft attention. + --cp-kv-cache-interleave-size "${cp_kv_cache_interleave_size}" + --dcp-kv-cache-interleave-size "${cp_kv_cache_interleave_size}" + --max-num-seqs "${max_num_seqs}" + --max-model-len "${max_model_len}" + --max-num-batched-tokens "${max_num_batched_tokens}" + --max-cudagraph-capture-size "${max_cudagraph_capture_size}" + --gpu-memory-utilization "${gpu_memory_utilization}" + --mamba-cache-mode align + --enable-chunked-prefill + --dtype bfloat16 + --kv-cache-dtype "${kv_cache_dtype}" + --quantization modelopt_mixed + --block-size 256 + --load-format "${load_format}" + --attention-backend "${attention_backend}" + --moe-backend "${moe_backend}" + --linear-backend "${linear_backend}" + --no-enable-flashinfer-autotune + --enable-auto-tool-choice + --tool-call-parser glm47 + --reasoning-parser glm45 + --additional-config + "{\"glm53_kda_decode_backend\":\"${kda_decode_backend}\",\"kda_prefill_backend\":\"${kda_prefill_backend}\"}" + --compilation-config + "{\"cudagraph_mode\":\"${cudagraph_mode}\"}" +) + +if ((chat_defaults_from_cli == 0)); then + # Preserve reasoning at its original assistant turn for agent continuations. + # Request template kwargs remain authoritative over these server defaults. + cmd+=(--default-chat-template-kwargs '{"reasoning_effort":"high","clear_thinking":false}') +fi + +if ((generation_defaults_from_cli == 0)); then + # The quantized checkpoint can omit Z.ai's generation defaults. Request + # sampling remains authoritative; an explicit CLI configuration replaces + # this launcher preset without creating duplicate JSON options. + cmd+=(--override-generation-config '{"temperature":1.0,"top_p":0.95}') +fi + +if ((prefill_interval_from_cli == 0)); then + cmd+=(--prefill-schedule-interval "${prefill_schedule_interval}") +fi + +if ((b12x_pcie_allreduce == 0)); then + cmd+=(--disable-custom-all-reduce) +fi + +case "${ENABLE_PREFIX_CACHING:-1}" in + 0) ;; + 1) cmd+=(--enable-prefix-caching) ;; + *) + printf 'ENABLE_PREFIX_CACHING must be 0 or 1; got %s\n' \ + "${ENABLE_PREFIX_CACHING}" >&2 + exit 2 + ;; +esac + +if ((num_speculative_tokens > 0)); then + case "${speculator}" in + mtp) + if [[ -n "${model_revision}" && "${model}" != /* ]]; then + cmd+=( + --speculative-config + "{\"method\":\"mtp\",\"revision\":\"${model_revision}\",\"num_speculative_tokens\":${num_speculative_tokens},\"draft_sample_method\":\"probabilistic\",\"rejection_sample_method\":\"standard\",\"moe_backend\":\"${mtp_moe_backend}\",\"attention_backend\":\"${mtp_attention_backend}\"}" + ) + else + cmd+=( + --speculative-config + "{\"method\":\"mtp\",\"num_speculative_tokens\":${num_speculative_tokens},\"draft_sample_method\":\"probabilistic\",\"rejection_sample_method\":\"standard\",\"moe_backend\":\"${mtp_moe_backend}\",\"attention_backend\":\"${mtp_attention_backend}\"}" + ) + fi + ;; + dflash | dflash2) + if [[ -n "${dflash_model_revision}" && "${dflash_model}" != /* ]]; then + cmd+=( + --speculative-config + "{\"method\":\"dflash\",\"model\":\"${dflash_model}\",\"revision\":\"${dflash_model_revision}\",\"num_speculative_tokens\":${num_speculative_tokens},\"draft_sample_method\":\"probabilistic\",\"rejection_sample_method\":\"standard\",\"kv_cache_dtype\":\"${dflash_kv_cache_dtype}\",\"attention_backend\":\"${dflash_attention_backend}\"}" + ) + else + cmd+=( + --speculative-config + "{\"method\":\"dflash\",\"model\":\"${dflash_model}\",\"num_speculative_tokens\":${num_speculative_tokens},\"draft_sample_method\":\"probabilistic\",\"rejection_sample_method\":\"standard\",\"kv_cache_dtype\":\"${dflash_kv_cache_dtype}\",\"attention_backend\":\"${dflash_attention_backend}\"}" + ) + fi + ;; + esac +fi + +cmd+=("$@") + +if [[ "${DRY_RUN:-0}" == 1 ]]; then + printf 'GLM-5.3-Flash NVFP4 launch:' + printf ' %q' "${cmd[@]}" + printf '\n' + exit 0 +fi + +exec "${cmd[@]}" diff --git a/recipes/glm53/source_locked_image_labels.py b/recipes/glm53/source_locked_image_labels.py new file mode 100644 index 00000000..814699a1 --- /dev/null +++ b/recipes/glm53/source_locked_image_labels.py @@ -0,0 +1,176 @@ +"""Derive serving-image metadata from authenticated build inputs. + +Runtime foundation labels may describe a different serving source tree. Clear +those inherited claims while preserving metadata for unchanged dependencies. +The source lock identifies implementation; qualification belongs to the evidence +report for the resulting immutable image, not to a pre-build assertion. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import sys +from pathlib import Path + +DEPENDENCY_PREFIXES = tuple( + f"local-inference.{name}." + for name in ( + "cuda", + "cutlass-dsl", + "deepgemm", + "exllamav3", + "flash-attention", + "flashinfer", + "instanttensor", + "nccl", + "nccl4py", + "pytorch", + "torch", + "torchvision", + "rust", + "transformers", + "xgrammar", + ) +) + + +def read_lock(path: Path) -> dict[str, str]: + values = {} + for line in path.read_text().splitlines(): + key, value = line.split("=", 1) + if key in values: + raise ValueError(f"Duplicate source-lock key: {key}") + values[key] = value + return values + + +def image_labels( + lock: dict[str, str], inherited: dict[str, str], source_lock_sha256: str +) -> dict[str, str]: + """Replace inherited serving claims without changing executable inputs.""" + labels = { + key: "" + for key in inherited + if ( + key.startswith("local-inference.") + and not key.startswith(DEPENDENCY_PREFIXES) + ) + or key.startswith("org.opencontainers.image.") + } + labels.update( + { + "org.opencontainers.image.title": "Jovian Judgement community serving", + "org.opencontainers.image.description": "GLM-5.3-Flash, Qwen3.8-Flash-Next and DeepSeek-V4-Flash serving; complete Git sources and model-specific launch profiles", + "org.opencontainers.image.version": lock["release.version"], + "org.opencontainers.image.source": "https://github.com/local-inference-lab/blackwell-llm-docker", + "org.opencontainers.image.revision": lock["vllm.commit"], + "local-inference.status": "implemented", + "local-inference.release.name": lock["release.name"], + "local-inference.release.rootfs-format": "flattened runtime plus source installation", + "local-inference.release.overlay2-rootfs-layers": lock[ + "runtime.rootfs.layers" + ], + "local-inference.runtime.base-image": lock["runtime.base.image"], + "local-inference.runtime.source-lock.path": "/opt/glm53-flash/source.lock", + "local-inference.runtime.source-lock.sha256": source_lock_sha256, + "local-inference.runtime.source-mode": "complete Git bundles", + "local-inference.runtime.cache-fingerprint": lock[ + "runtime.cache.fingerprint" + ], + "local-inference.runtime.cudagraph-mode": lock["runtime.cudagraph.mode"], + "local-inference.runtime.default.moe-backend": lock.get( + "runtime.moe-backend.default", "auto" + ), + "local-inference.runtime.default.gpu-local.target-page-tokens": "2048", + "local-inference.runtime.default.recurrent-checkpoint-policy": "auto", + "local-inference.runtime.lmcache-transfer": lock[ + "runtime.lmcache.transfer" + ], + "local-inference.scheduler.max-num-batched-tokens": lock[ + "runtime.scheduler.max-num-batched-tokens" + ], + "local-inference.scheduler.prefill-schedule-interval": "1", + "local-inference.scheduler.prefill-compute-share": lock[ + "runtime.scheduler.prefill-compute-share" + ], + "local-inference.model.repository": lock["model.repository"], + "local-inference.model.update-policy": "resolve requested Hugging Face revision at startup", + "local-inference.draft.model": lock["draft.repository"], + "local-inference.draft.quantization": lock["draft.quantization"], + "local-inference.draft.update-policy": "resolve requested Hugging Face revision at startup", + "local-inference.glm53.mtp-backends": "attention:B12X,moe:Marlin; private NVFP4 vocabulary head", + "local-inference.glm53.target-backends": "attention:B12X,moe:B12X,linear:B12X", + "local-inference.backend.gdn.prefill": "FlashKDA with recurrent checkpoints", + "local-inference.backend.gdn.decode": "B12X", + "local-inference.backend.allreduce": "B12X PCIe with NCCL for larger transfers", + "local-inference.lmcache.version": lock["lmcache.version"], + "com.nvidia.cuda.version": lock["runtime.cuda.version"], + "com.nvidia.pytorch.version": lock["runtime.pytorch.version"], + } + ) + for name in ("vllm", "b12x", "lmcache"): + for key, source_key in ( + ("commit", "commit"), + ("source-revision", "commit"), + ("tree", "tree"), + ("package-tree", "package.tree"), + ): + labels[f"local-inference.{name}.{key}"] = lock[f"{name}.{source_key}"] + labels[f"local-inference.{name}.repo"] = lock.get(f"{name}.repository", "") + labels[f"local-inference.{name}.branch"] = "image-source" + if "vllm.version" in lock: + labels["local-inference.vllm.version"] = lock["vllm.version"] + for key in ("base.commit", "patch.sha256", "extension.sha256"): + labels[f"local-inference.flashkda.{key}"] = lock[f"flashkda.{key}"] + if "flashinfer.commit" in lock: + labels.update( + { + key: "" + for key in inherited + if key.startswith("local-inference.flashinfer.") + } + ) + for key in ( + "commit", + "artifact.image", + "python.wheel.sha256", + "jit-cache.wheel.sha256", + ): + labels[f"local-inference.flashinfer.{key}"] = lock[f"flashinfer.{key}"] + labels["local-inference.flashinfer.version"] = "0.6.18+cu133" + labels["local-inference.flashinfer.repo"] = ( + "https://github.com/voipmonitor/flashinfer.git" + ) + if "vllm.native.extension.sha256" in lock: + labels["local-inference.vllm.native.extension.sha256"] = lock[ + "vllm.native.extension.sha256" + ] + labels["local-inference.ds4.entrypoint"] = "/usr/local/bin/serve-ds4-jovian.sh" + labels["local-inference.ds4.target-backends"] = ( + "attention:B12X,moe:B12X-W4A8,linear:DeepGEMM" + ) + return labels + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--source-lock", type=Path, required=True) + args = parser.parse_args() + base = json.load(sys.stdin)[0] + labels = image_labels( + read_lock(args.source_lock), + base["Config"].get("Labels") or {}, + hashlib.sha256(args.source_lock.read_bytes()).hexdigest(), + ) + arguments = [] + for key, value in sorted(labels.items()): + if "\0" in key or "\0" in value: + raise ValueError("Image labels must not contain NUL bytes") + arguments.append(f"{key}={value}\0".encode()) + sys.stdout.buffer.write(b"".join(arguments)) + + +if __name__ == "__main__": + main() diff --git a/recipes/glm53/tests/test_build_contract.py b/recipes/glm53/tests/test_build_contract.py new file mode 100644 index 00000000..05d8fc24 --- /dev/null +++ b/recipes/glm53/tests/test_build_contract.py @@ -0,0 +1,232 @@ +"""Authenticate installed source and metadata while preserving native dependencies.""" + +import importlib.util +import subprocess +from pathlib import Path + +import pytest + +RECIPE = Path(__file__).resolve().parents[1] +spec = importlib.util.spec_from_file_location( + "source_locked_image_labels", RECIPE / "source_locked_image_labels.py" +) +labels = importlib.util.module_from_spec(spec) +spec.loader.exec_module(labels) +native_spec = importlib.util.spec_from_file_location( + "build_jovian_stable_native", RECIPE / "build_jovian_stable_native.py" +) +native = importlib.util.module_from_spec(native_spec) +native_spec.loader.exec_module(native) +bundle_spec = importlib.util.spec_from_file_location( + "prepare_glm53_source_bundles", RECIPE / "prepare_glm53_source_bundles.py" +) +bundler = importlib.util.module_from_spec(bundle_spec) +bundle_spec.loader.exec_module(bundler) + + +@pytest.fixture +def lock(): + values = { + "release.name": "serving-fixture", + "release.version": "test", + "runtime.base.image": "runtime@sha256:" + "a" * 64, + "runtime.rootfs.layers": "2", + "runtime.cache.fingerprint": "source-fixture", + "runtime.cudagraph.mode": "FULL_AND_PIECEWISE", + "runtime.lmcache.transfer": "engine-driven asynchronous shared memory", + "runtime.scheduler.max-num-batched-tokens": "4096", + "runtime.scheduler.prefill-compute-share": "0.4", + "runtime.cuda.version": "13.3", + "runtime.pytorch.version": "2.13.0", + "model.repository": "target/fixture", + "draft.repository": "draft/fixture", + "draft.quantization": "MXFP8", + "lmcache.version": "0.5.5.dev0", + } + for name in ("vllm", "b12x", "lmcache"): + for key in ("commit", "tree", "package.tree"): + values[f"{name}.{key}"] = f"{name}-{key}" + for key in ("base.commit", "patch.sha256", "extension.sha256"): + values[f"flashkda.{key}"] = key + return values + + +def test_serving_labels_replace_incompatible_foundation_claims(lock): + inherited = { + "local-inference.vllm.composition": "different PR stack", + "local-inference.cache.target-block-size": "512", + "local-inference.glm53.mtp-backends": "moe:Humming", + "local-inference.runtime.source-lock.sha256": "different-lock", + "local-inference.scheduler.prefill-schedule-interval": "8", + "org.opencontainers.image.version": "different-image", + } + overrides = labels.image_labels(lock, inherited, "b" * 64) + assert overrides["local-inference.vllm.composition"] == "" + assert overrides["local-inference.cache.target-block-size"] == "" + assert "moe:Marlin" in overrides["local-inference.glm53.mtp-backends"] + assert overrides["local-inference.runtime.source-lock.sha256"] == "b" * 64 + assert overrides["local-inference.scheduler.prefill-schedule-interval"] == "1" + assert overrides["org.opencontainers.image.version"] == "test" + assert overrides["local-inference.status"] == "implemented" + assert all("qualified" not in value for value in overrides.values()) + + +def test_unchanged_dependency_labels_are_preserved_by_inheritance(lock): + dependency = "local-inference.flashinfer.commit" + overrides = labels.image_labels(lock, {dependency: "source-ref"}, "b" * 64) + assert dependency not in overrides + for name in ("vllm", "b12x", "lmcache"): + assert overrides[f"local-inference.{name}.commit"] == lock[f"{name}.commit"] + + +def test_deployment_moe_backend_is_declared_in_image_and_lock(lock): + lock["runtime.moe-backend.default"] = "b12x" + overrides = labels.image_labels(lock, {}, "b" * 64) + assert overrides["local-inference.runtime.default.moe-backend"] == "b12x" + dockerfile = (RECIPE / "Dockerfile.glm53-cache-contracts").read_text() + assert "VLLM_DEFAULT_MOE_BACKEND=b12x" in dockerfile + + +def test_source_repository_labels_use_manifest_not_inferred_ownership(lock): + lock["vllm.repository"] = "https://github.com/voipmonitor/vllm.git" + overrides = labels.image_labels(lock, {}, "b" * 64) + assert overrides["local-inference.vllm.repo"] == lock["vllm.repository"] + assert overrides["local-inference.b12x.repo"] == "" + + +@pytest.mark.parametrize( + "url", + [ + "https://name:secret@example.org/repo.git", + "/tmp/repo", + "https://example.org/repo?secret=value", + ], +) +def test_bundle_metadata_rejects_credentials_and_local_paths(url): + with pytest.raises(ValueError, match="HTTPS URL without credentials"): + bundler.repository_url(url) + + +def test_bundle_metadata_normalizes_github_ssh_url(): + assert ( + bundler.repository_url("git@github.com:voipmonitor/vllm.git") + == "https://github.com/voipmonitor/vllm.git" + ) + + +def test_replaced_flashinfer_artifact_does_not_inherit_another_source_ref(lock): + for key in ( + "commit", + "artifact.image", + "python.wheel.sha256", + "jit-cache.wheel.sha256", + ): + lock[f"flashinfer.{key}"] = key + overrides = labels.image_labels( + lock, {"local-inference.flashinfer.ref": "unrelated-source"}, "b" * 64 + ) + assert overrides["local-inference.flashinfer.ref"] == "" + assert overrides["local-inference.flashinfer.commit"] == "commit" + assert overrides["local-inference.flashinfer.artifact.image"] == "artifact.image" + + +def test_ambiguous_source_lock_is_rejected(tmp_path): + source = tmp_path / "source.lock" + source.write_text("vllm.commit=a\nvllm.commit=b\n") + with pytest.raises(ValueError, match="Duplicate"): + labels.read_lock(source) + + +def git(directory: Path, *arguments: str) -> str: + return subprocess.check_output( + ["git", "-C", str(directory), *arguments], text=True + ).strip() + + +@pytest.fixture +def source_bundle(tmp_path: Path): + source = tmp_path / "source" + source.mkdir() + git(source, "init", "--quiet") + git(source, "config", "user.name", "Source fixture") + git(source, "config", "user.email", "source@example.invalid") + (source / ".gitignore").write_text("*.so\n") + (source / "module.py").write_text("value = 1\n") + (source / "obsolete.py").write_text("value = 2\n") + git(source, "add", ".") + git(source, "commit", "--quiet", "-m", "Source fixture") + destination = tmp_path / "installed source" + subprocess.run( + ["git", "clone", "--quiet", str(source), str(destination)], check=True + ) + (destination / "extension.so").write_bytes(b"qualified native artifact") + (source / "obsolete.py").unlink() + (source / "module.py").write_text("value = 3\n") + git(source, "add", "-A") + git(source, "commit", "--quiet", "-m", "Complete source fixture") + commit = git(source, "rev-parse", "HEAD") + tree = git(source, "rev-parse", "HEAD^{tree}") + bundle = tmp_path / "source.bundle" + git(source, "bundle", "create", str(bundle), "HEAD") + return destination, bundle, commit, tree + + +def test_installed_tree_matches_bundle_without_losing_native(source_bundle): + destination, bundle, commit, tree = source_bundle + for _ in range(2): + subprocess.run( + [ + "bash", + str(RECIPE / "install_source_bundle.sh"), + str(bundle), + commit, + tree, + str(destination), + ], + check=True, + ) + assert git(destination, "rev-parse", "HEAD") == commit + assert git(destination, "status", "--porcelain") == "" + assert (destination / "module.py").read_text() == "value = 3\n" + assert not (destination / "obsolete.py").exists() + assert ( + destination / "extension.so" + ).read_bytes() == b"qualified native artifact" + + +def test_cutlass_identity_records_the_actual_committed_tree(source_bundle): + destination, _, commit, tree = source_bundle + assert native.cutlass_identity(destination) == { + "cutlass.commit": git(destination, "rev-parse", "HEAD"), + "cutlass.tree": git(destination, "rev-parse", "HEAD^{tree}"), + } + (destination / "module.py").write_text("uncommitted modification\n") + with pytest.raises(ValueError, match="committed CUTLASS"): + native.cutlass_identity(destination) + + +@pytest.mark.parametrize("mismatch", ["commit", "tree"]) +def test_lock_mismatch_does_not_replace_installed_sources(source_bundle, mismatch): + destination, bundle, commit, tree = source_bundle + before = git(destination, "rev-parse", "HEAD") + if mismatch == "commit": + commit = "0" * 40 + else: + tree = "0" * 40 + result = subprocess.run( + [ + "bash", + str(RECIPE / "install_source_bundle.sh"), + str(bundle), + commit, + tree, + str(destination), + ], + capture_output=True, + text=True, + check=False, + ) + assert result.returncode != 0 + assert git(destination, "rev-parse", "HEAD") == before + assert (destination / "module.py").read_text() == "value = 1\n" + assert (destination / "obsolete.py").read_text() == "value = 2\n" diff --git a/recipes/glm53/tests/test_ds4_sampling_defaults.py b/recipes/glm53/tests/test_ds4_sampling_defaults.py new file mode 100644 index 00000000..6af95a00 --- /dev/null +++ b/recipes/glm53/tests/test_ds4_sampling_defaults.py @@ -0,0 +1,60 @@ +"""Check DS4 launcher sampling precedence without loading model weights.""" + +import json +import os +import shlex +import subprocess +from pathlib import Path + +import pytest + +RECIPE = Path(__file__).resolve().parents[1] + + +@pytest.fixture +def launch(tmp_path): + stub = tmp_path / "argv.sh" + stub.write_text('#!/bin/bash\nprintf "%s\\0" "$@"\n') + stub.chmod(0o755) + wrapper = tmp_path / "serve.sh" + source = (RECIPE / "serve-ds4-jovian.sh").read_text() + wrapper.write_text( + source.replace( + "exec /usr/local/bin/lmcache-mp-wrapper.sh", + f"exec {shlex.quote(str(stub))}", + ) + ) + + def run(arguments=(), environment=None): + result = subprocess.run( + ["bash", str(wrapper), *arguments], + env={"PATH": os.environ["PATH"], **(environment or {})}, + capture_output=True, + text=True, + check=True, + ) + return result.stdout.rstrip("\0").split("\0")[1:] + + return run + + +@pytest.mark.parametrize("variant", ["text", "vision"]) +def test_agent_profile_uses_publisher_sampling(launch, variant): + arguments = launch(environment={"DS4_MODEL_VARIANT": variant}) + assert arguments[0] == "--override-generation-config" + assert json.loads(arguments[1]) == {"temperature": 1.0, "top_p": 0.95} + + +@pytest.mark.parametrize( + "arguments,environment", + [ + (["--generation-config", "vllm"], {}), + (["--override-generation-config.top_p=0.8"], {}), + (['--override-generation-config={"top_p":0.8}'], {}), + (["--config", "/operator/serve.yaml"], {}), + ([], {"EXTRA_VLLM_ARGS": '--override-generation-config={"top_p":0.8}'}), + ([], {"EXTRA_VLLM_ARGS": "--generation_config vllm"}), + ], +) +def test_operator_sampling_is_authoritative(launch, arguments, environment): + assert launch(arguments, environment) == arguments diff --git a/recipes/glm53/tests/test_lmcache_http_launcher.py b/recipes/glm53/tests/test_lmcache_http_launcher.py new file mode 100644 index 00000000..62523bb0 --- /dev/null +++ b/recipes/glm53/tests/test_lmcache_http_launcher.py @@ -0,0 +1,406 @@ +"""Validate sidecar bind arguments without starting a cache or model server.""" + +import json +import os +import shlex +import subprocess +import sys +from pathlib import Path + +import pytest + +RECIPE = Path(__file__).resolve().parents[1] +WRAPPERS = ( + "serve-glm53-flash-lmcache.sh", + "serve-glm53-flash-lmcache-cache-complete.sh", +) + + +def render(wrapper, host=None, dtype=None, settings=None, cwd=None): + source = (RECIPE / wrapper).read_text() + # Use the test environment's interpreter for JSON validation; no LMCache + # or vLLM package is imported while rendering arguments. + source = source.replace("/opt/venv/bin/python", shlex.quote(sys.executable)) + # The configuration section validates settings and constructs argv. Stop + # before child-process supervision; no sidecar or vLLM process is launched. + configuration, separator, _ = source.partition("lmcache_pid=\n") + assert separator + if dtype is not None: + # The standalone wrapper resolves vLLM's dtype after defining its + # supervisor functions, but before broker or child-process creation. + configuration, separator, _ = source.partition( + "if [[ ${transfer_mode} != engine_driven ]]; then" + ) + assert separator + environment = { + "PATH": os.environ["PATH"], + "LMCACHE_ENABLED": "1", + "LMCACHE_L2_ENABLED": "0", + "LMCACHE_MIN_SHM_GIB": "1", + } + if host is not None: + environment["LMCACHE_HTTP_HOST"] = host + if dtype is not None: + environment["LMCACHE_KV_CACHE_DTYPE"] = dtype + environment.update(settings or {}) + return subprocess.run( + [ + "bash", + "-c", + configuration + + '\nprintf "%s\\0" "${health_url}" "${lmcache_server[@]}" "vllm_dtype=${KV_CACHE_DTYPE:-}"', + ], + env=environment, + cwd=cwd, + capture_output=True, + text=True, + check=False, + ) + + +@pytest.mark.parametrize("wrapper", WRAPPERS) +@pytest.mark.parametrize("transfer", ["engine_driven", "lmcache_driven", "auto"]) +def test_engine_transport_has_a_stable_shared_memory_arena(wrapper, transfer): + result = render( + wrapper, + settings={ + "LMCACHE_TRANSFER_MODE": transfer, + "LMCACHE_INSTANCE_ID": "glm/test", + "LMCACHE_MP_PORT": "5566", + }, + ) + assert result.returncode == 0, result.stderr + fields = result.stdout.rstrip("\0").split("\0") + if transfer == "engine_driven": + assert fields.count("--no-l1-use-lazy") == 1 + assert "--l1-use-lazy" not in fields + assert fields.count("--shm-name") == 1 + assert fields[fields.index("--shm-name") + 1] == "lmcache-glm_test-5566" + else: + assert fields.count("--l1-use-lazy") == 1 + assert "--no-l1-use-lazy" not in fields + assert "--shm-name" not in fields + + +@pytest.mark.parametrize("wrapper", WRAPPERS) +def test_engine_transport_preserves_explicit_shared_memory_name(wrapper): + result = render( + wrapper, + settings={ + "LMCACHE_TRANSFER_MODE": "engine_driven", + "LMCACHE_SHM_NAME": "glm-checkpoints.1", + }, + ) + assert result.returncode == 0, result.stderr + fields = result.stdout.rstrip("\0").split("\0") + assert fields[fields.index("--shm-name") + 1] == "glm-checkpoints.1" + + +@pytest.mark.parametrize("wrapper", WRAPPERS) +@pytest.mark.parametrize("name", ["a/b", "a b", "$(id)"]) +def test_invalid_shared_memory_name_fails_before_server_start(wrapper, name): + result = render( + wrapper, + settings={"LMCACHE_TRANSFER_MODE": "engine_driven", "LMCACHE_SHM_NAME": name}, + ) + assert result.returncode == 2 + assert "LMCACHE_SHM_NAME" in result.stderr + + +@pytest.mark.parametrize("wrapper", WRAPPERS) +@pytest.mark.parametrize("policy", [None, "default", "retain"]) +def test_prefetch_retention_defaults_to_reusable_ram(wrapper, policy): + settings = {} if policy is None else {"LMCACHE_L2_PREFETCH_POLICY": policy} + result = render(wrapper, settings=settings) + assert result.returncode == 0, result.stderr + fields = result.stdout.rstrip("\0").split("\0") + assert fields.count("--l2-prefetch-policy") == 1 + assert fields[fields.index("--l2-prefetch-policy") + 1] == (policy or "retain") + + +@pytest.mark.parametrize("wrapper", WRAPPERS) +@pytest.mark.parametrize("policy", [None, "default", "retain"]) +@pytest.mark.parametrize("l2_enabled", ["0", "1"]) +def test_retained_filesystem_restore_can_reclaim_unowned_ram( + wrapper, policy, l2_enabled, tmp_path +): + settings = {"LMCACHE_L2_ENABLED": l2_enabled, "LMCACHE_L2_PATH": str(tmp_path)} + if policy is not None: + settings["LMCACHE_L2_PREFETCH_POLICY"] = policy + result = render(wrapper, settings=settings) + assert result.returncode == 0, result.stderr + fields = result.stdout.rstrip("\0").split("\0") + expected = l2_enabled == "1" and policy != "default" + assert fields.count("--emergency-evict-for-prefetch") == int(expected) + assert "--write-back-on-evict" not in fields + + +@pytest.mark.parametrize("wrapper", WRAPPERS) +@pytest.mark.parametrize("policy", ["all", "Retain", "retain default"]) +def test_invalid_retention_policy_fails_before_server_start(wrapper, policy): + result = render(wrapper, settings={"LMCACHE_L2_PREFETCH_POLICY": policy}) + assert result.returncode == 2 + assert "LMCACHE_L2_PREFETCH_POLICY" in result.stderr + + +@pytest.mark.parametrize("wrapper", WRAPPERS) +def test_server_arguments_preserve_literal_wildcards_without_execution( + wrapper, tmp_path +): + (tmp_path / "match.yaml").touch() + result = render( + wrapper, + settings={ + "LMCACHE_SERVER_EXTRA_ARGS": "--max-cpu-workers 4 --pattern *.yaml $(touch${IFS}EXECUTED)" + }, + cwd=tmp_path, + ) + assert result.returncode == 0, result.stderr + fields = result.stdout.rstrip("\0").split("\0") + assert fields[-6:-1] == [ + "--max-cpu-workers", + "4", + "--pattern", + "*.yaml", + "$(touch${IFS}EXECUTED)", + ] + assert not (tmp_path / "EXECUTED").exists() + + +@pytest.mark.parametrize("wrapper", WRAPPERS) +@pytest.mark.parametrize( + "argument", + [ + "--shm-name other", + "--chunk-size=512", + "--no-separate-object-groups", + "--http-port 9999", + "--l2-prefetch-policy=default", + "--no-emergency-evict-for-prefetch", + ], +) +def test_extra_arguments_cannot_override_transfer_or_readiness_contract( + wrapper, argument +): + result = render(wrapper, settings={"LMCACHE_SERVER_EXTRA_ARGS": argument}) + assert result.returncode == 2 + assert "launcher-managed option" in result.stderr + + +@pytest.mark.parametrize("wrapper", WRAPPERS) +@pytest.mark.parametrize( + "arguments", ["--max-cpu-workers 4\n--max-workers 8", "--max-workers 4\r"] +) +def test_extra_arguments_reject_silently_truncated_lines(wrapper, arguments): + result = render(wrapper, settings={"LMCACHE_SERVER_EXTRA_ARGS": arguments}) + assert result.returncode == 2 + assert "one whitespace-separated line" in result.stderr + + +@pytest.mark.parametrize("wrapper", WRAPPERS) +@pytest.mark.parametrize( + "host,expected,probe,warning", + [ + (None, "127.0.0.1", "127.0.0.1", False), + ("0.0.0.0", "0.0.0.0", "127.0.0.1", True), + ("192.0.2.10", "192.0.2.10", "192.0.2.10", True), + ("::", "::", "[::1]", True), + ("::1", "::1", "[::1]", False), + ("cache.internal", "cache.internal", "cache.internal", True), + ("127.example.internal", "127.example.internal", "127.example.internal", True), + ("127.0.0.2", "127.0.0.2", "127.0.0.2", False), + ("127.999.0.1", "127.999.0.1", "127.999.0.1", True), + ], +) +def test_http_bind_and_readiness_use_compatible_addresses( + wrapper, host, expected, probe, warning +): + result = render(wrapper, host) + assert result.returncode == 0, result.stderr + fields = result.stdout.rstrip("\0").split("\0") + assert fields[0] == f"http://{probe}:8085/healthcheck" + assert fields.count("--http-host") == 1 + assert fields[fields.index("--http-host") + 1] == expected + assert ("administrative APIs" in result.stderr) == warning + + +@pytest.mark.parametrize("wrapper", WRAPPERS) +@pytest.mark.parametrize( + "host", ["--host", "localhost;id", "a b", "$(id)", "a/b", "a\nb"] +) +def test_invalid_bind_is_rejected_before_process_start(wrapper, host): + result = render(wrapper, host) + assert result.returncode == 2 + assert "LMCACHE_HTTP_HOST" in result.stderr + + +@pytest.mark.parametrize( + "dtype,expected", + [ + ("fp8_ds_mla", "fp8"), + ("fp8", "fp8"), + ("fp8_e4m3", "fp8_e4m3"), + ("nvfp4_ds_mla", "nvfp4_ds_mla"), + ], +) +def test_standalone_cache_storage_dtype_maps_to_vllm(dtype, expected): + result = render("serve-glm53-flash-lmcache.sh", dtype=dtype) + assert result.returncode == 0, result.stderr + assert result.stdout.rstrip("\0").split("\0")[-1] == f"vllm_dtype={expected}" + + +@pytest.mark.parametrize( + "transfer,resolved", + [("engine_driven", "request_boundaries"), ("lmcache_driven", "aligned")], +) +@pytest.mark.parametrize( + "policy_args", + [ + [], + ["--recurrent-checkpoint-policy", "auto"], + ["--recurrent-checkpoint-policy=auto"], + ], +) +def test_cache_policy_is_resolved_once_before_delegation( + transfer, resolved, policy_args +): + unrelated = ["--max-model-len", "65536", "--served-model-name", "model with spaces"] + result = subprocess.run( + [ + "bash", + str(RECIPE / "serve-glm53-flash-cache-complete.sh"), + *policy_args, + *unrelated, + ], + env={ + "PATH": os.environ["PATH"], + "CACHE_CONFIG_DRY_RUN": "1", + "CACHE_MODE": "lmcache", + "LMCACHE_TRANSFER_MODE": transfer, + }, + capture_output=True, + text=True, + check=False, + ) + assert result.returncode == 0, result.stderr + line = next(line for line in result.stdout.splitlines() if line.startswith("ARGV:")) + args = shlex.split(line.removeprefix("ARGV:")) + assert args == unrelated + ["--recurrent-checkpoint-policy", resolved] + + +@pytest.mark.parametrize( + "args", + [ + [], + ["--recurrent-checkpoint-policy", "request_boundaries"], + ["--recurrent-checkpoint-policy=request_boundaries"], + ], +) +def test_semantic_sidecar_does_not_duplicate_native_checkpoint_policy(args): + source = (RECIPE / "serve-glm53-flash-lmcache-cache-complete.sh").read_text() + configuration, separator, _ = source.partition("lmcache_pid=\n") + assert separator + _, separator, connector = source.partition("connector_config=$(printf") + assert separator + connector, separator, _ = connector.partition('"${base_launcher}" "$@"') + assert separator + script = ( + configuration + + "\nvllm_extra_args=()\nconnector_config=$(printf" + + connector + + '\nprintf "%s\\0" "$@" "${vllm_extra_args[@]}"' + ) + result = subprocess.run( + ["bash", "-c", script, "policy-test", *args], + env={ + "PATH": os.environ["PATH"], + "LMCACHE_ENABLED": "1", + "LMCACHE_L2_ENABLED": "0", + "LMCACHE_MIN_SHM_GIB": "1", + "LMCACHE_TRANSFER_MODE": "engine_driven", + "LMCACHE_CHECKPOINT_IDENTITY": '{"target_revision":"test"}', + }, + capture_output=True, + text=True, + check=False, + ) + assert result.returncode == 0, result.stderr + fields = result.stdout.rstrip("\0").split("\0") + assert ( + sum( + field.split("=", 1)[0] == "--recurrent-checkpoint-policy" + for field in fields + ) + == 1 + ) + + +@pytest.mark.parametrize("policy", ["aligned", "request_boundaries"]) +@pytest.mark.parametrize("target_identity", ["a" * 64, "c" * 64]) +def test_persistent_namespace_uses_effective_model_and_content_identity( + policy, target_identity +): + source = (RECIPE / "serve-glm53-flash-cache-complete.sh").read_text() + resolver = ( + "/opt/venv/bin/python \\\n" + " /usr/local/libexec/glm53_checkpoint_identity.py" + ) + assert source.count(resolver) == 1 + source = source.replace(resolver, "identity_fixture") + invocation = '\nexec "${cache_launcher}" "$@"' + assert source.count(invocation) == 1 + source = source.replace( + invocation, + 'printf "%s\\0" "${LMCACHE_L2_PATH}" ' + '"${LMCACHE_CHECKPOINT_IDENTITY:-}" "${MODEL_REVISION}" "$@"', + ) + identity = { + "checkpoint_identity": { + "target_revision": target_identity, + "source_revision": "b" * 64, + "draft_revision": "", + }, + "model_revision": "", + "draft_model_revision": "", + } + fixture = ( + 'identity_fixture() { printf "%s\\0" "$@" >&2; ' + + "printf '%s' " + + shlex.quote(json.dumps(identity)) + + "; }\n" + ) + result = subprocess.run( + [ + "bash", + "-c", + fixture + source, + "cache-launcher-test", + "/models/alternate", + "--recurrent-checkpoint-policy", + policy, + ], + env={ + "PATH": os.environ["PATH"], + "CACHE_MODE": "lmcache", + "LMCACHE_TRANSFER_MODE": "engine_driven", + "LMCACHE_L2_ENABLED": "1", + "MODEL": "unused/repository", + "MTP_DEPTH": "0", + }, + capture_output=True, + text=True, + check=False, + ) + assert result.returncode == 0, result.stderr + identity_args = result.stderr.rstrip("\0").split("\0") + assert identity_args[identity_args.index("--model") + 1] == "/models/alternate" + fields = result.stdout.rstrip("\0").split("\0") + assert f"/_models_alternate-{target_identity}/" in fields[0] + assert "unused" not in fields[0] and "huggingface-main" not in fields[0] + assert bool(fields[1]) == (policy == "request_boundaries") + assert fields[2] == "" + assert fields[3:] == [ + "/models/alternate", + "--recurrent-checkpoint-policy", + policy, + ] diff --git a/recipes/glm53/tests/test_lmcache_native_reuse.py b/recipes/glm53/tests/test_lmcache_native_reuse.py new file mode 100644 index 00000000..bc38ee1d --- /dev/null +++ b/recipes/glm53/tests/test_lmcache_native_reuse.py @@ -0,0 +1,196 @@ +"""Verify native reuse rejects source drift and produces an installable wheel.""" + +import base64 +import csv +from email.parser import BytesParser +import hashlib +import importlib.util +import io +from pathlib import Path +import subprocess +import sys +import zipfile + +import pytest + +HELPER = Path(__file__).resolve().parents[1] / "pack_lmcache_native_reuse.py" +SPEC = importlib.util.spec_from_file_location("lmcache_native_reuse", HELPER) +packer = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(packer) +TAG = "cp312-cp312-linux_x86_64" + + +@pytest.fixture +def wheel_members(): + return { + "lmcache/__init__.py": b'"""Python package fixture."""\n', + "lmcache-0.1.dist-info/WHEEL": ( + b"Wheel-Version: 1.0\nRoot-Is-Purelib: true\nTag: py3-none-any\n\n" + ), + "lmcache-0.1.dist-info/METADATA": b"Name: lmcache\nVersion: 0.1\n", + "lmcache-0.1.dist-info/RECORD": b"obsolete record\n", + } + + +def test_platform_metadata_and_complete_record_preserve_payloads(wheel_members): + native = {"lmcache/cuda_ops.so": b"CUDA fixture", "lmcache/nested/rust.so": b"Rust"} + packed = packer.platform_wheel(wheel_members, native, TAG) + for name, data in native.items(): + assert packed[name] == data + assert packed["lmcache/__init__.py"] == wheel_members["lmcache/__init__.py"] + metadata = packed["lmcache-0.1.dist-info/WHEEL"].decode() + assert "Root-Is-Purelib: false" in metadata + assert metadata.count("Tag:") == 1 + assert f"Tag: {TAG}" in metadata + headers = BytesParser().parsebytes(packed["lmcache-0.1.dist-info/WHEEL"]) + assert headers.get_all("Tag") == [TAG] + assert headers["Root-Is-Purelib"] == "false" + rows = list( + csv.reader(io.StringIO(packed["lmcache-0.1.dist-info/RECORD"].decode())) + ) + assert {row[0] for row in rows} == set(packed) + for name, digest, size in rows: + if name.endswith("/RECORD"): + assert (digest, size) == ("", "") + else: + expected = base64.urlsafe_b64encode(hashlib.sha256(packed[name]).digest()) + assert digest == "sha256=" + expected.rstrip(b"=").decode() + assert size == str(len(packed[name])) + + +@pytest.mark.parametrize( + "native,tag", + [ + ({}, TAG), + ({"outside.so": b"bad"}, TAG), + ({"lmcache/a.so": b"a"}, "py3-none-any"), + ], +) +def test_incompatible_native_payload_is_rejected(wheel_members, native, tag): + with pytest.raises(ValueError): + packer.platform_wheel(wheel_members, native, tag) + + +def test_cuda_reuse_preserves_rebuilt_cpu_libraries(wheel_members): + rebuilt = { + f"lmcache/{name}.so": b"rebuilt " + name.encode() + for name in packer.CPU_EXTENSIONS + } + reference = {f"lmcache/{name}.so": b"reference" for name in packer.CPU_EXTENSIONS} + reference["lmcache/cuda_ops.so"] = b"qualified CUDA" + files = {**wheel_members, **rebuilt} + selected = packer.select_native(files, reference, "cpu-rebuild-cuda-reuse") + result = packer.platform_wheel(files, selected, TAG) + assert result["lmcache/cuda_ops.so"] == b"qualified CUDA" + for path, data in rebuilt.items(): + assert result[path] == data + del files["lmcache/lmcache_fs.so"] + with pytest.raises(ValueError, match="all three rebuilt CPU"): + packer.select_native(files, reference, "cpu-rebuild-cuda-reuse") + + +@pytest.mark.parametrize("mode", ["reuse-all", "cpu-rebuild-cuda-reuse"]) +@pytest.mark.parametrize( + "changed", + [ + None, + "csrc/storage_backends/fs/connector.cpp", + "csrc/cuda/kernel.cu", + "csrc/shared.h", + "setup.py", + ], +) +def test_cli_reuses_only_native_inputs_with_identical_git_identity( + tmp_path, wheel_members, mode, changed +): + reference = tmp_path / "reference" + reference.mkdir() + paths = [name for name in packer.NATIVE_INPUTS if name != "csrc"] + paths += [ + "csrc/storage_backends/fs/connector.cpp", + "csrc/cuda/kernel.cu", + "csrc/shared.h", + ] + for name in paths: + (reference / name).parent.mkdir(parents=True, exist_ok=True) + (reference / name).write_text(f"native input: {name}\n") + subprocess.run(["git", "init", "-q", str(reference)], check=True) + subprocess.run(["git", "-C", str(reference), "add", "."], check=True) + git_commit = [ + "git", + "-c", + "user.name=Fixture", + "-c", + "user.email=fixture@example.invalid", + "commit", + "-qm", + "Declare native inputs", + ] + subprocess.run(git_commit, cwd=reference, check=True) + source = tmp_path / "source" + subprocess.run(["git", "clone", "-q", str(reference), str(source)], check=True) + (source / "python.py").write_text("# Python-only source change\n") + if changed: + (source / changed).write_text("changed native source\n") + subprocess.run(["git", "add", "."], cwd=source, check=True) + subprocess.run(git_commit, cwd=source, check=True) + package = tmp_path / "site-packages" / "lmcache" + package.mkdir(parents=True) + (package / "cuda_ops.so").write_bytes(b"compiled CUDA fixture") + for name in packer.CPU_EXTENSIONS: + (package / f"{name}.so").write_bytes(b"reference CPU fixture") + dist_info = package.parent / "lmcache-0.1.dist-info" + dist_info.mkdir() + (dist_info / "WHEEL").write_text(f"Tag: {TAG}\n") + wheel_dir = tmp_path / "wheels" + wheel_dir.mkdir() + suffix = "py3-none-any" if mode == "reuse-all" else TAG + pure = wheel_dir / f"lmcache-0.1-{suffix}.whl" + if mode == "cpu-rebuild-cuda-reuse": + wheel_members.update( + { + f"lmcache/{name}.so": b"rebuilt CPU fixture" + for name in packer.CPU_EXTENSIONS + } + ) + with zipfile.ZipFile(pure, "w") as archive: + for name, data in wheel_members.items(): + archive.writestr(name, data) + result = subprocess.run( + [ + sys.executable, + str(HELPER), + "--source", + str(source), + "--reference-source", + str(reference), + "--reference-package", + str(package), + "--wheel-dir", + str(wheel_dir), + "--mode", + mode, + ], + capture_output=True, + text=True, + check=False, + ) + platform = wheel_dir / f"lmcache-0.1-{TAG}.whl" + compatible = changed is None or ( + mode == "cpu-rebuild-cuda-reuse" + and changed == "csrc/storage_backends/fs/connector.cpp" + ) + if not compatible: + assert result.returncode != 0 + assert "native build inputs differ" in result.stderr + assert pure.exists() + if pure != platform: + assert not platform.exists() + else: + assert result.returncode == 0, result.stderr + if pure != platform: + assert not pure.exists() + with zipfile.ZipFile(platform) as archive: + assert archive.read("lmcache/cuda_ops.so") == b"compiled CUDA fixture" + if mode == "cpu-rebuild-cuda-reuse": + assert archive.read("lmcache/lmcache_fs.so") == b"rebuilt CPU fixture" diff --git a/recipes/glm53/tests/test_scheduler_launcher.py b/recipes/glm53/tests/test_scheduler_launcher.py new file mode 100644 index 00000000..65e1fba8 --- /dev/null +++ b/recipes/glm53/tests/test_scheduler_launcher.py @@ -0,0 +1,263 @@ +"""Validate scheduler environment mapping without loading models or using GPUs.""" + +import json +import os +import shlex +import subprocess +from pathlib import Path + +import pytest + +RECIPE = Path(__file__).resolve().parents[1] +OPTIONS = { + "PREFILL_COMPUTE_SHARE": "--prefill-compute-share", + "PREFILL_COMPUTE_HALF_LIFE": "--prefill-compute-half-life", + "MAX_PARALLEL_PREFILLS": "--max-parallel-prefills", + "PREFILL_POLICY": "--prefill-policy", + "DECODE_REFILL_TARGET": "--decode-refill-target", +} + + +@pytest.fixture +def launch(tmp_path): + stub = tmp_path / "argv.sh" + stub.write_text('#!/bin/bash\nprintf "%s\\0" "$@"\n') + stub.chmod(0o755) + wrapper = tmp_path / "scheduler.sh" + source = (RECIPE / "serve-glm53-flash-nvfp4-dflash2-scheduler-qos.sh").read_text() + wrapper.write_text( + source.replace( + "readonly base_launcher=/usr/local/libexec/serve-glm53-flash-nvfp4-dflash2.sh", + f"readonly base_launcher={shlex.quote(str(stub))}", + ) + ) + + def run(environment=None, arguments=(), *, base=False): + env = { + "PATH": os.environ["PATH"], + "CUDAGRAPH_CAPTURE_SIZES": "none", + "PREFILL_SCHEDULE_INTERVAL": "1", + **(environment or {}), + } + path = wrapper + if base: + path = RECIPE / "serve-glm53-flash-nvfp4-dflash2.sh" + env["DRY_RUN"] = "1" + return subprocess.run( + ["bash", str(path), *arguments], + env=env, + capture_output=True, + text=True, + check=False, + ) + + return run + + +def argv(result): + assert result.returncode == 0, result.stderr + return [value for value in result.stdout.split("\0") if value] + + +def test_absent_controls_preserve_native_defaults(launch): + assert argv(launch()) == [] + + +def test_community_defaults_do_not_enable_interleaving(launch): + assert argv( + launch({"FAIRNESS_ENGINE": "compute_share", "PREFILL_COMPUTE_SHARE": "0.4"}) + ) == [ + "--prefill-compute-share", + "0.4", + ] + + +def test_glm_sampling_uses_publisher_defaults_without_checkpoint_metadata(launch): + result = launch(base=True) + assert result.returncode == 0, result.stderr + arguments = shlex.split(result.stdout)[3:] + option = "--override-generation-config" + assert arguments.count(option) == 1 + assert json.loads(arguments[arguments.index(option) + 1]) == { + "temperature": 1.0, + "top_p": 0.95, + } + + +@pytest.mark.parametrize( + "arguments", + [ + ["--override-generation-config", '{"top_p":0.8}'], + ['--override-generation-config={"top_p":0.8}'], + ["--override-generation-config.top_p", "0.8"], + ["--override_generation_config.top_p=0.8"], + ["--generation-config", "vllm"], + ["--generation-config=/operator/model-defaults"], + ["--config", "/operator/serve.yaml"], + ], +) +def test_explicit_generation_policy_is_not_overwritten(launch, arguments): + result = launch(arguments=arguments, base=True) + assert result.returncode == 0, result.stderr + rendered = shlex.split(result.stdout)[3:] + assert rendered[-len(arguments) :] == arguments + assert '{"temperature":1.0,"top_p":0.95}' not in rendered + + +def test_all_environment_controls_are_forwarded_once(launch): + values = dict(zip(OPTIONS, ["auto", "responsive", "auto", "decode-aware", "auto"])) + arguments = argv(launch(values)) + for environment, option in OPTIONS.items(): + assert arguments.count(option) == 1 + assert arguments[arguments.index(option) + 1] == values[environment] + + +@pytest.mark.parametrize("equals", [False, True]) +def test_cli_overrides_invalid_environment_values(launch, equals): + values = dict(zip(OPTIONS.values(), ["auto", "smooth", "4", "round-robin", "2"])) + arguments = [] + for option, value in values.items(): + arguments.extend([f"{option}={value}"] if equals else [option, value]) + assert argv(launch({name: "invalid" for name in OPTIONS}, arguments)) == arguments + + +@pytest.mark.parametrize( + "half_life", ["smooth", "responsive", "0.5", "2", "1e-3", "+2."] +) +def test_auto_half_life(launch, half_life): + assert "--prefill-compute-half-life" in argv( + launch( + { + "PREFILL_COMPUTE_SHARE": "auto", + "PREFILL_COMPUTE_HALF_LIFE": half_life, + } + ) + ) + + +@pytest.mark.parametrize( + "name,value", + [ + ("PREFILL_COMPUTE_SHARE", "0"), + ("PREFILL_COMPUTE_SHARE", "1"), + ("PREFILL_COMPUTE_SHARE", "nan"), + ("PREFILL_COMPUTE_SHARE", "0.4junk"), + ("PREFILL_COMPUTE_SHARE", "-0.4"), + ("PREFILL_COMPUTE_HALF_LIFE", "0"), + ("PREFILL_COMPUTE_HALF_LIFE", "inf"), + ("PREFILL_COMPUTE_HALF_LIFE", "1e9999"), + ("PREFILL_COMPUTE_HALF_LIFE", "garbage"), + ("MAX_PARALLEL_PREFILLS", "0"), + ("MAX_PARALLEL_PREFILLS", "1.5"), + ("DECODE_REFILL_TARGET", "-1"), + ("PREFILL_POLICY", "fastest"), + ], +) +def test_invalid_controls_fail_before_model_launch(launch, name, value): + result = launch({"PREFILL_COMPUTE_SHARE": "auto", name: value}) + assert result.returncode == 2 + assert name in result.stderr + + +def test_numeric_share_with_half_life_fails(launch): + assert ( + launch( + {"PREFILL_COMPUTE_SHARE": "0.4", "PREFILL_COMPUTE_HALF_LIFE": "smooth"} + ).returncode + == 2 + ) + + +def test_cli_effective_share_controls_half_life_validation(launch): + assert ( + launch( + {"PREFILL_COMPUTE_SHARE": "auto", "PREFILL_COMPUTE_HALF_LIFE": "smooth"}, + ["--prefill-compute-share", "0.4"], + ).returncode + == 2 + ) + + +def test_compatibility_selector_and_explicit_cli(launch): + environment = {"FAIRNESS_ENGINE": "none", "PREFILL_COMPUTE_SHARE": "0.4"} + assert argv(launch(environment)) == [] + assert argv(launch(environment, ["--prefill-compute-share", "auto"])) == [ + "--prefill-compute-share", + "auto", + ] + assert launch({"FAIRNESS_ENGINE": "compute_share"}).returncode == 2 + assert launch({"FAIRNESS_ENGINE": "micro_slicing"}).returncode == 2 + + +@pytest.mark.parametrize( + "option", list(OPTIONS.values()) + ["--prefill-schedule-interval"] +) +def test_duplicate_and_missing_cli_values_fail(launch, option): + assert launch(arguments=[option, "auto", f"{option}=auto"]).returncode == 2 + assert launch(arguments=[option]).returncode == 2 + + +def test_fairness_interval_uses_cli_precedence(launch): + environment = { + "PREFILL_COMPUTE_SHARE": "0.4", + "PREFILL_SCHEDULE_INTERVAL": "invalid", + } + arguments = ["--prefill-schedule-interval=1"] + assert argv(launch(environment, arguments)) == arguments + [ + "--prefill-compute-share", + "0.4", + ] + assert launch(environment).returncode == 2 + assert ( + launch( + {"PREFILL_COMPUTE_SHARE": "0.4"}, ["--prefill-schedule-interval", "8"] + ).returncode + == 2 + ) + + +def test_base_interval_is_not_duplicated_and_chat_defaults_are_explicit(launch): + result = launch( + {"PREFILL_SCHEDULE_INTERVAL": "invalid"}, + ["--prefill-schedule-interval", "1"], + base=True, + ) + assert result.returncode == 0, result.stderr + arguments = shlex.split(result.stdout.split("launch:", 1)[1]) + assert arguments.count("--prefill-schedule-interval") == 1 + assert json.loads( + arguments[arguments.index("--default-chat-template-kwargs") + 1] + ) == {"reasoning_effort": "high", "clear_thinking": False} + + +@pytest.mark.parametrize("equals", [False, True]) +def test_explicit_chat_defaults_are_forwarded_without_a_second_object(launch, equals): + value = '{"reasoning_effort":"max","clear_thinking":true}' + option = "--default-chat-template-kwargs" + supplied = [f"{option}={value}"] if equals else [option, value] + result = launch(arguments=supplied, base=True) + assert result.returncode == 0, result.stderr + arguments = shlex.split(result.stdout.split("launch:", 1)[1]) + settings = [item for item in arguments if item.split("=", 1)[0] == option] + assert len(settings) == 1 + if equals: + assert settings == [f"{option}={value}"] + else: + assert arguments[arguments.index(option) + 1] == value + + +def test_graph_capture_controls_are_preserved(launch): + assert argv(launch({"CUDAGRAPH_CAPTURE_SIZES": "1 4 8"})) == [ + "--cudagraph-capture-sizes", + "1", + "4", + "8", + ] + + +def test_help_describes_all_scheduler_controls(launch): + result = launch(arguments=["--help"]) + assert result.returncode == 0 + for environment, option in OPTIONS.items(): + assert environment in result.stdout + assert option in result.stdout diff --git a/recipes/glm53/tests/test_source_version.py b/recipes/glm53/tests/test_source_version.py new file mode 100644 index 00000000..67fd3bc2 --- /dev/null +++ b/recipes/glm53/tests/test_source_version.py @@ -0,0 +1,51 @@ +"""Generated package versions must identify installed serving Python sources.""" + +import importlib.util +from pathlib import Path + +import pytest + +RECIPE = Path(__file__).resolve().parents[1] +spec = importlib.util.spec_from_file_location( + "install_vllm_source_version", RECIPE / "install_vllm_source_version.py" +) +installer = importlib.util.module_from_spec(spec) +spec.loader.exec_module(installer) + + +@pytest.fixture +def source(tmp_path): + (tmp_path / "vllm").mkdir() + (tmp_path / "vllm.egg-info").mkdir() + (tmp_path / "vllm.egg-info/PKG-INFO").write_text( + "Name: vllm\nVersion: 0.0.0\nSummary: Serving library\n" + ) + return tmp_path + + +def test_python_and_distribution_versions_agree(source): + version = "0.26.1rc0+glm53.dspark.vllmabcdef01" + installer.install_version(source, version) + namespace = {} + exec((source / "vllm/_version.py").read_text(), namespace) + assert namespace["__version__"] == version + assert namespace["version"] == version + assert namespace["version_tuple"] == (0, 26, 1, "rc0", "glm53.dspark.vllmabcdef01") + assert (source / "vllm.egg-info/PKG-INFO").read_text() == ( + f"Name: vllm\nVersion: {version}\nSummary: Serving library\n" + ) + + +def test_ambiguous_distribution_metadata_is_rejected_before_writing(source): + metadata = source / "vllm.egg-info/PKG-INFO" + metadata.write_text("Version: 1\nVersion: 2\n") + with pytest.raises(ValueError, match="exactly one"): + installer.install_version(source, "0.26.1rc0+glm53.dspark") + assert not (source / "vllm/_version.py").exists() + assert metadata.read_text() == "Version: 1\nVersion: 2\n" + + +def test_unidentified_source_version_is_rejected(source): + with pytest.raises(ValueError, match="source identity"): + installer.install_version(source, "0.26.1rc0") + assert not (source / "vllm/_version.py").exists() diff --git a/scripts/compose_vllm_release.py b/scripts/compose_vllm_release.py old mode 100644 new mode 100755 index ec68f96b..e6bd01c3 --- a/scripts/compose_vllm_release.py +++ b/scripts/compose_vllm_release.py @@ -13,7 +13,6 @@ from pathlib import Path from typing import Any - SHA_RE = re.compile(r"^[0-9a-f]{40}$") SHA256_RE = re.compile(r"^[0-9a-f]{64}$") @@ -35,8 +34,7 @@ def _run( env=env, check=False, text=True, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, + capture_output=True, ) if check and completed.returncode: detail = completed.stderr.strip() or completed.stdout.strip() @@ -83,6 +81,13 @@ def _load_manifest(path: Path) -> dict[str, Any]: ) manifest["composition_strategy"] = strategy + for key in ("base_commit", "expected_tree"): + value = manifest.get(key) + if value is not None and ( + not isinstance(value, str) or not SHA_RE.fullmatch(value) + ): + raise CompositionError(f"{key} must be a 40-character Git object ID") + composition_ref = manifest.get("composition_ref") composition_commit = manifest.get("composition_commit") if strategy == "pinned_branch": @@ -132,6 +137,13 @@ def _load_manifest(path: Path) -> dict[str, Any]: ) entry["ref"] = ref entry["commits"] = commits + requires = entry.get("requires", []) + if not isinstance(requires, list) or any( + type(dependency) is not int or dependency <= 0 for dependency in requires + ): + raise CompositionError(f"PR #{number} requires must be a list of PR numbers") + if not set(requires).issubset(seen): + raise CompositionError(f"PR #{number} has unresolved review dependencies") seen.add(number) source_patches = manifest.get("source_patches", []) @@ -154,6 +166,87 @@ def _load_manifest(path: Path) -> dict[str, Any]: return manifest +def verify_review_tree(manifest_path: Path, repository: Path) -> dict[str, Any]: + """Prove pinned review heads reproduce a tree without editing a checkout. + + The object database must contain the audited public commits. This offline + check proves composition, not publication or PR membership; the release PR + audit must establish those separately. + """ + manifest = _load_manifest(manifest_path) + if manifest["composition_strategy"] != "merge" or manifest["source_patches"]: + raise CompositionError( + "Review-tree verification requires merges without source patches" + ) + if not manifest.get("base_commit") or not manifest.get("expected_tree"): + raise CompositionError( + "Review-tree verification requires base_commit and expected_tree" + ) + tip = manifest["base_commit"] + env = os.environ.copy() + env.update( + { + "GIT_AUTHOR_NAME": "Release Composer", + "GIT_AUTHOR_EMAIL": "release-composer@local", + "GIT_COMMITTER_NAME": "Release Composer", + "GIT_COMMITTER_EMAIL": "release-composer@local", + "GIT_AUTHOR_DATE": "2000-01-01T00:00:00+00:00", + "GIT_COMMITTER_DATE": "2000-01-01T00:00:00+00:00", + } + ) + resolved = [] + _run(["git", "cat-file", "-e", f"{tip}^{{commit}}"], cwd=repository) + for entry in manifest["pull_requests"]: + head = entry["head"] + _run(["git", "cat-file", "-e", f"{head}^{{commit}}"], cwd=repository) + merged = _run( + ["git", "merge-tree", "--write-tree", tip, head], + cwd=repository, + check=False, + ) + if merged.returncode: + raise CompositionError( + f"PR #{entry['number']} conflicts with its declared predecessors. " + "Publish the resolution in a review head before composing.\n" + + merged.stdout + + merged.stderr + ) + tree = merged.stdout.splitlines()[0] + tip = _run( + [ + "git", + "commit-tree", + tree, + "-p", + tip, + "-p", + head, + "-m", + f"Compose reviewed source: PR #{entry['number']}", + ], + cwd=repository, + env=env, + ).stdout.strip() + resolved.append({"number": entry["number"], "head": head, "tree": tree}) + tree = _run(["git", "rev-parse", f"{tip}^{{tree}}"], cwd=repository).stdout.strip() + if tree != manifest["expected_tree"]: + detail = _run( + ["git", "diff", "--stat", manifest["expected_tree"], tree], + cwd=repository, + ).stdout + raise CompositionError( + f"Composed tree {tree} differs from expected tree " + f"{manifest['expected_tree']}.\n{detail}" + ) + return { + "base_commit": manifest["base_commit"], + "review_units": resolved, + "result_commit": tip, + "result_tree": tree, + "matches_expected_tree": True, + } + + def compose(manifest_path: Path, output_dir: Path) -> dict[str, Any]: manifest_path = manifest_path.resolve() output_dir = output_dir.resolve() @@ -163,7 +256,7 @@ def compose(manifest_path: Path, output_dir: Path) -> dict[str, Any]: composition_strategy = str(manifest["composition_strategy"]) composition_ref = manifest.get("composition_ref") composition_commit = manifest.get("composition_commit") - base_sha = _remote_sha(repository, base_ref) + base_sha = manifest.get("base_commit") or _remote_sha(repository, base_ref) output_dir.mkdir(parents=True, exist_ok=True) patch_path = output_dir / "integration.patch" @@ -173,7 +266,7 @@ def compose(manifest_path: Path, output_dir: Path) -> dict[str, Any]: checkout = Path(temp) / "checkout" _run(["git", "init", "--quiet", str(checkout)]) _run(["git", "remote", "add", "origin", repository], cwd=checkout) - _run(["git", "fetch", "--quiet", "--no-tags", "origin", base_ref], cwd=checkout) + _run(["git", "fetch", "--quiet", "--no-tags", "origin", base_sha], cwd=checkout) _run(["git", "checkout", "--quiet", "--detach", base_sha], cwd=checkout) merge_env = os.environ.copy() @@ -382,8 +475,18 @@ def compose(manifest_path: Path, output_dir: Path) -> dict[str, Any]: } ) - _run(["git", "diff", "--check", base_sha], cwd=checkout) + # Unified-diff blobs encode empty context lines as a literal space. + # Preserve that syntax; whitespace checks still cover source files. + _run( + ["git", "diff", "--check", base_sha, "--", ".", ":(exclude)*.patch"], + cwd=checkout, + ) result_tree = _run(["git", "write-tree"], cwd=checkout).stdout.strip() + if manifest.get("expected_tree") not in (None, result_tree): + raise CompositionError( + f"Composed tree {result_tree} differs from expected tree " + f"{manifest['expected_tree']}" + ) patch = subprocess.run( ["git", "diff", "--binary", "--full-index", base_sha], cwd=checkout, @@ -432,10 +535,20 @@ def compose(manifest_path: Path, output_dir: Path) -> dict[str, Any]: def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("manifest", type=Path) - parser.add_argument("--output-dir", required=True, type=Path) + parser.add_argument("--output-dir", type=Path) + parser.add_argument( + "--verify-in", + type=Path, + help="Verify pinned review trees in an existing Git object database without checkout edits", + ) args = parser.parse_args() try: - lock = compose(args.manifest, args.output_dir) + if args.verify_in is not None: + lock = verify_review_tree(args.manifest, args.verify_in) + else: + if args.output_dir is None: + parser.error("--output-dir is required unless --verify-in is used") + lock = compose(args.manifest, args.output_dir) except CompositionError as exc: parser.error(str(exc)) print(json.dumps(lock, sort_keys=True)) diff --git a/tests/test_compose_vllm_release.py b/tests/test_compose_vllm_release.py index 226986c3..9473ff7c 100644 --- a/tests/test_compose_vllm_release.py +++ b/tests/test_compose_vllm_release.py @@ -7,8 +7,7 @@ import pytest -from scripts.compose_vllm_release import CompositionError, compose - +from scripts.compose_vllm_release import CompositionError, compose, verify_review_tree REPOSITORY_ROOT = Path(__file__).resolve().parents[1] @@ -19,8 +18,7 @@ def _git(cwd: Path, *args: str) -> str: cwd=cwd, check=True, text=True, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, + capture_output=True, ).stdout.strip() @@ -128,6 +126,23 @@ def test_composition_lock_is_reproducible(tmp_path: Path) -> None: ).read_bytes() +def test_composition_preserves_unified_diff_context_whitespace(tmp_path: Path) -> None: + source, remote, _ = _repositories(tmp_path) + patch = "--- a/model.py\n+++ b/model.py\n@@ -1,2 +1,3 @@\n first\n \n+last\n" + head = _create_pr(source, remote, 1, "kernel.patch", patch) + manifest = _manifest(tmp_path / "manifest.json", remote, [(1, head)]) + lock = compose(manifest, tmp_path / "output") + assert lock["result"]["tree"] == _git(source, "rev-parse", f"{head}^{{tree}}") + + +def test_composition_rejects_source_trailing_whitespace(tmp_path: Path) -> None: + source, remote, _ = _repositories(tmp_path) + head = _create_pr(source, remote, 1, "feature.py", "value = 1 \n") + manifest = _manifest(tmp_path / "manifest.json", remote, [(1, head)]) + with pytest.raises(CompositionError, match="trailing whitespace"): + compose(manifest, tmp_path / "output") + + def test_composition_supports_a_base_without_pull_requests(tmp_path: Path) -> None: _, remote, base = _repositories(tmp_path) manifest = _manifest(tmp_path / "manifest.json", remote, []) @@ -141,6 +156,106 @@ def test_composition_supports_a_base_without_pull_requests(tmp_path: Path) -> No assert (output / "integration.patch").read_bytes() == b"" +def _pin_tree(path: Path, base: str, tree: str) -> Path: + manifest = json.loads(path.read_text()) + manifest.update(base_commit=base, expected_tree=tree) + path.write_text(json.dumps(manifest)) + return path + + +def test_review_tree_preserves_authors_and_dirty_checkout(tmp_path: Path) -> None: + source, remote, base = _repositories(tmp_path) + head = _create_pr(source, remote, 1, "feature.py", "feature\n") + manifest = _pin_tree( + _manifest(tmp_path / "manifest.json", remote, [(1, head)]), + base, + _git(source, "rev-parse", f"{head}^{{tree}}"), + ) + (source / "model.py").write_text("user edits\n") + _git(source, "add", "model.py") + index_before = _git(source, "write-tree") + first = verify_review_tree(manifest, source) + second = verify_review_tree(manifest, source) + assert first == second + assert first["matches_expected_tree"] + assert _git(source, "write-tree") == index_before + assert _git(source, "rev-parse", "HEAD") == base + assert (source / "model.py").read_text() == "user edits\n" + assert _git(source, "rev-parse", f"{first['result_commit']}^2") == head + assert ( + _git(source, "show", "-s", "--format=%an <%ae>", head) + == "Test " + ) + + +def test_review_tree_rejects_missing_source_changes(tmp_path: Path) -> None: + source, remote, base = _repositories(tmp_path) + head = _create_pr(source, remote, 1, "feature.py", "feature\n") + manifest = _pin_tree( + _manifest(tmp_path / "manifest.json", remote, []), + base, + _git(source, "rev-parse", f"{head}^{{tree}}"), + ) + with pytest.raises(CompositionError, match="differs from expected tree"): + verify_review_tree(manifest, source) + + +def test_review_tree_requires_public_conflict_resolution(tmp_path: Path) -> None: + source, remote, base = _repositories(tmp_path) + first = _create_pr(source, remote, 1, "model.py", "first\n") + second = _create_pr(source, remote, 2, "model.py", "second\n") + manifest = _pin_tree( + _manifest(tmp_path / "manifest.json", remote, [(1, first), (2, second)]), + base, + _git(source, "rev-parse", f"{base}^{{tree}}"), + ) + with pytest.raises(CompositionError, match="PR #2 conflicts"): + verify_review_tree(manifest, source) + assert _git(source, "status", "--porcelain") == "" + + +def test_review_tree_requires_ordered_dependencies(tmp_path: Path) -> None: + source, remote, base = _repositories(tmp_path) + head = _create_pr(source, remote, 1, "feature.py", "feature\n") + path = _pin_tree( + _manifest(tmp_path / "manifest.json", remote, [(1, head)]), + base, + _git(source, "rev-parse", f"{head}^{{tree}}"), + ) + manifest = json.loads(path.read_text()) + manifest["pull_requests"][0]["requires"] = [2] + path.write_text(json.dumps(manifest)) + with pytest.raises(CompositionError, match="unresolved review dependencies"): + verify_review_tree(path, source) + + +def test_composition_pins_base_when_branch_advances(tmp_path: Path) -> None: + source, remote, base = _repositories(tmp_path) + tree = _git(source, "rev-parse", f"{base}^{{tree}}") + _commit(source, "base-update.py", "advanced\n", "advance base") + _git(source, "push", "--quiet", "origin", "dev/gilded-gnosis") + manifest = _pin_tree(_manifest(tmp_path / "manifest.json", remote, []), base, tree) + lock = compose(manifest, tmp_path / "output") + assert lock["base"]["commit"] == base + assert lock["result"]["tree"] == tree + + +@pytest.mark.parametrize("requires", [None, "1", [True], [-1], [{}]]) +def test_review_tree_rejects_invalid_dependencies(tmp_path: Path, requires) -> None: + source, remote, base = _repositories(tmp_path) + head = _create_pr(source, remote, 1, "feature.py", "feature\n") + path = _pin_tree( + _manifest(tmp_path / "manifest.json", remote, [(1, head)]), + base, + _git(source, "rev-parse", f"{head}^{{tree}}"), + ) + manifest = json.loads(path.read_text()) + manifest["pull_requests"][0]["requires"] = requires + path.write_text(json.dumps(manifest)) + with pytest.raises(CompositionError, match="requires must be a list"): + verify_review_tree(path, source) + + def test_composition_applies_digest_locked_source_patch(tmp_path: Path) -> None: source, remote, base = _repositories(tmp_path) (source / "model.py").write_text("patched runtime\n") diff --git a/tests/test_glm53_reasoning_default.py b/tests/test_glm53_reasoning_default.py new file mode 100644 index 00000000..744ff7e6 --- /dev/null +++ b/tests/test_glm53_reasoning_default.py @@ -0,0 +1,58 @@ +"""GLM launcher reasoning defaults without model loading or CUDA access.""" + +import json +import os +import shlex +import subprocess +import unittest +from pathlib import Path + +LAUNCHER = ( + Path(__file__).resolve().parents[1] + / "recipes/glm53/serve-glm53-flash-nvfp4-dflash2.sh" +) + + +class ReasoningDefaultTest(unittest.TestCase): + def command(self, mode, *args): + env = dict(os.environ, DRY_RUN="1", SPECULATOR=mode, MTP_DEPTH="3") + result = subprocess.run( + ["bash", str(LAUNCHER), *args], + env=env, + check=True, + capture_output=True, + text=True, + ) + return shlex.split(result.stdout.split(" launch:", 1)[1]) + + def defaults(self, command): + # argparse uses the last complete occurrence of the option, allowing + # operator-supplied JSON to override launcher defaults. + values = [ + json.loads(command[i + 1]) + for i, arg in enumerate(command) + if arg == "--default-chat-template-kwargs" + ] + self.assertTrue(values) + return values[-1] + + def test_high_for_both_speculator_families(self): + for mode in ("mtp", "dflash2"): + with self.subTest(mode=mode): + self.assertEqual( + self.defaults(self.command(mode)), + {"reasoning_effort": "high", "clear_thinking": False}, + ) + + def test_operator_can_override_default_json(self): + override = {"reasoning_effort": "max", "clear_thinking": True} + self.assertEqual( + self.defaults( + self.command("mtp", "--default-chat-template-kwargs", json.dumps(override)) + ), + override, + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/verify_b12x_release_contracts.py b/tests/verify_b12x_release_contracts.py new file mode 100644 index 00000000..7b4c7d42 --- /dev/null +++ b/tests/verify_b12x_release_contracts.py @@ -0,0 +1,100 @@ +"""CPU-only graph-dispatch and quantization guards for the community image. + +Run with the image's Python and B12X package. FakeTensor dispatch must not +execute a CUDA implementation. Keyword arguments follow the installed operator +schema; extending that schema must not invalidate this graph-interface check. +These checks do not qualify CUDA arithmetic or serving performance. +""" + +import importlib + +import pytest +import torch +from torch._subclasses.fake_tensor import FakeTensorMode + + +@pytest.mark.parametrize( + "module,operation", + [ + ("b12x._lib.dense_gemm", "dense_gemm_launch"), + *[ + ("b12x.norm.mhc._kernels", operation) + for operation in ( + "mhc_pre_partial_launch", + "mhc_post_pre_partial_launch", + "mhc_finalize_gram_launch", + "mhc_prefill_tf32_project_launch", + ) + ], + *[ + ("b12x.moe.fused_moe._impl", operation) + for operation in ( + "tp_moe_dynamic_launch", + "tp_moe_compact_micro_launch", + ) + ], + *[ + ("b12x.moe._shared.kernels.w4a16.kernel", operation) + for operation in ( + "w4a16_small_m_direct_launch", + "w4a16_fused_moe_launch", + "w4a16_fused_moe_calibrated_launch", + "w4a16_topk_sum_launch", + ) + ], + ], +) +def test_launch_operator_fake_dispatch(module, operation): + importlib.import_module(module) + operator = getattr(torch.ops.b12x, operation).default + primitives = {"int": 1, "bool": False, "float": 1.0, "str": "default"} + assert not operator._schema.returns + with FakeTensorMode(): + operands = {} + for argument in operator._schema.arguments: + kind = str(argument.type) + if argument.has_default_value(): + operands[argument.name] = argument.default_value + elif kind == "Tensor": + operands[argument.name] = torch.empty((2, 128), dtype=torch.bfloat16) + elif kind in primitives: + operands[argument.name] = primitives[kind] + elif kind.startswith("Optional["): + operands[argument.name] = None + else: + raise AssertionError(f"Unsupported schema type {kind}: {argument.name}") + assert operator(**operands) is None + + +@pytest.mark.parametrize( + "quant_mode,intermediate_size,expected", + [ + ("nvfp4", 640, True), + ("nvfp4", 320, False), + ("w4a16", 640, False), + ("w4a8_mx", 640, False), + ], + ids=["qwen-target-tp1", "qwen-target-tp2", "qwen-draft", "mxfp4"], +) +def test_shared_scales_preserve_recipe_and_tile_guards( + monkeypatch, quant_mode, intermediate_size, expected +): + """An input-scale equality proof cannot override FP4 recipe/tile limits.""" + from b12x.moe.fused_moe import _impl + + monkeypatch.setenv("B12X_NVFP4_DYNAMIC_MATERIALIZED", "1") + monkeypatch.setenv("B12X_DYNAMIC_WORK_SOURCE", "persistent_grid") + assert ( + _impl._nvfp4_dynamic_materialized_enabled( + quant_mode=quant_mode, + activation="silu", + routed_rows=4096 * 10, + num_experts=512, + k=2560, + n=intermediate_size, + share_input_across_experts=True, + deterministic_output=False, + planned_tile_m=128, + ) + is expected + ) diff --git a/tests/verify_glm53_reasoning_default.py b/tests/verify_glm53_reasoning_default.py new file mode 100644 index 00000000..a8d13a1f --- /dev/null +++ b/tests/verify_glm53_reasoning_default.py @@ -0,0 +1,45 @@ +"""Verify GLM reasoning rendering with installed vLLM and real tokenizer files. + +Run inside a GLM serving image with CUDA_VISIBLE_DEVICES empty. This checks +request/default precedence and the exact model-visible prompt; it does not +infer reasoning behavior from output length. +""" + +import argparse +import json + +from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest +from vllm.tokenizers import get_tokenizer + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--model", required=True) + args = parser.parse_args() + tokenizer = get_tokenizer(args.model) + rows = [] + for label, defaults, extra, expected in ( + ("checkpoint_template_default", {}, {}, "Max"), + ("server_default", {"reasoning_effort": "high"}, {}, "High"), + ("explicit_max", {"reasoning_effort": "high"}, {"reasoning_effort": "max"}, "Max"), + ("explicit_high", {"reasoning_effort": "high"}, {"reasoning_effort": "high"}, "High"), + ("explicit_low", {"reasoning_effort": "high"}, {"reasoning_effort": "low"}, "Low"), + ("template_max", {"reasoning_effort": "high"}, + {"chat_template_kwargs": {"reasoning_effort": "max"}}, "Max"), + ): + request = ChatCompletionRequest( + model="GLM-5.3-Flash-NVFP4", + messages=[{"role": "user", "content": "What is 2 + 2?"}], + **extra, + ) + params = request.build_chat_params(None, "auto").with_defaults(defaults) + rendered = tokenizer.apply_chat_template( + request.messages, tokenize=False, **params.get_apply_chat_template_kwargs() + ) + assert rendered.startswith(f"[gMASK]<|system|>Reasoning Effort: {expected}<|user|>"), label + rows.append({"case": label, "effort": expected.lower(), "prompt": rendered}) + print(json.dumps({"status": "passed", "cases": rows}, indent=2)) + + +if __name__ == "__main__": + main()