Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
31 commits
Select commit Hold shift + click to select a range
5ffb5a4
exl3: restore serial prefill for mixed Trellis
malaiwah Jul 31, 2026
ddd3b31
exl3: bound opt-in Trellis prefill capacity
voipmonitor Jul 30, 2026
2b3583d
exl3: share prefill capacity validation
voipmonitor Jul 30, 2026
8b5cf23
test(exl3): cover bounded serial mixed prefill
voipmonitor Aug 1, 2026
0904272
[GG] quant: add EXL3 BF16 MXFP8 overlay
voipmonitor Aug 1, 2026
6681ae4
quant: reject mixed EXL3 packed overlays
voipmonitor Aug 1, 2026
2197da2
feat(exl3): load shared hidden-side rotations
voipmonitor Aug 1, 2026
2845a11
test(exl3): cover shared rotations in mixed tiers
voipmonitor Aug 1, 2026
a91c91d
moe: keep mixed EXL3 prefill on one grid
yatesdr Aug 1, 2026
e0c12c3
moe: restore stock one-grid prefill geometry
yatesdr Aug 1, 2026
9a60e4b
perf(exl3): select qualified block-32 mixed prefill
voipmonitor Aug 2, 2026
af87e64
feat(exl3): cache online K6 dense overlays
voipmonitor Aug 2, 2026
c8836a5
fix(exl3): harden online cache and runtime planning
voipmonitor Aug 2, 2026
5baeae2
fix(exl3): pass shared H layout to mixed Trellis
voipmonitor Aug 4, 2026
25e373c
perf(exl3): qualify block-32 for shared-H tier mixes
voipmonitor Aug 4, 2026
2ab1c1f
refactor(exl3): use renamed b12x runtime
voipmonitor Aug 7, 2026
155bf59
fix(exl3): prewarm mixed-Trellis route packing
malaiwah Aug 7, 2026
86fe19d
[Quant] EXL3: load R7 per-(expert, projection) trellis checkpoints
brandonmmusic-max Aug 9, 2026
cc8773d
[Quant] EXL3 R7: release ballast after load; skip MTP weights when MT…
brandonmmusic-max Aug 9, 2026
e874127
[Quant] EXL3 R7: keep projection-tight w13 with explicit gate counts
brandonmmusic-max Aug 9, 2026
e5d7de6
[Quant] EXL3 R7: isolate graph scratch, strict metadata, ABI-6 compat
Aug 9, 2026
bf0784a
[Docs] EXL3 R7: document _skip_disabled_mtp_weight
Aug 9, 2026
2f028c3
feat(exl3): qualify native R7 K3 K4 K5 runtime
voipmonitor Aug 10, 2026
654bdca
fix(exl3): harden online cache and MTP config handling
voipmonitor Aug 10, 2026
4e338fc
fix(exl3): summarize online overlay selection once
voipmonitor Aug 10, 2026
3c0a496
fix(exl3): bound cache waits and MTP layer lookup
voipmonitor Aug 10, 2026
8e7be4d
docs(exl3): explain mixed-bitrate runtime invariants
voipmonitor Aug 10, 2026
704d94e
[GG] perf(exl3): bound the B12X K6 route to a packed-N window; head a…
malaiwah Aug 17, 2026
8837c26
feat(exl3): online K-quant embedding table (VLLM_EXL3_EMBED_ONLINE_BITS)
malaiwah Aug 18, 2026
fd500ef
fix(exl3): chunked embed conversion, 0-row stub, init-time hook for o…
malaiwah Aug 19, 2026
3cd48af
fix: address CodeRabbit review — document cache identity limitation +…
malaiwah Aug 20, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 17 additions & 1 deletion docs/features/quantization/online.md
Original file line number Diff line number Diff line change
Expand Up @@ -100,7 +100,7 @@ vllm serve <glm-5.2-modelopt-checkpoint> \
Serialized checkpoint-quantized shared experts are preserved; only excluded
BF16 projection weights are converted at load time.

### MXFP8 dense linears on ModelOpt checkpoints
### MXFP8 dense linears on quantized checkpoints

The `linear` field overlays online MXFP8 the same way onto every other
BF16 dense linear the ModelOpt checkpoint excludes: attention projections,
Expand Down Expand Up @@ -143,6 +143,22 @@ vllm serve lukealonso/GLM-5.2-NVFP4 \
--quantization-config '{"linear":{"weight":"mxfp8"},"ignore":["re:.*kv_b_proj"]}'
```

EXL3 checkpoints can use the same overlay for dense and shared-expert
projections that are absent from the EXL3 tensor metadata. Serialized EXL3
dense matrices and routed experts always retain their checkpoint format;
`moe` overlays are rejected. `lm_head` also remains in its checkpoint format
or BF16. For example:

```bash
vllm serve <exl3-checkpoint> \
--quantization exl3 \
--quantization-config '{"linear":{"weight":"mxfp8"},"shared_experts":{"weight":"mxfp8"},"ignore":["re:.*\\.q_a_proj$","re:.*kv_a_proj_with_mqa","lm_head"]}'
```

As with ModelOpt, `linear` does not select shared experts. Add the
`shared_experts` field explicitly when those BF16 projections should also be
converted. Ignore patterns are matched against unfused checkpoint names.

### Activation overrides on already-quantized checkpoints

For checkpoint-quantized models, `quantization_config` lets you pick an
Expand Down
54 changes: 54 additions & 0 deletions tests/models/test_deepseek_v2_mtp_weights.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,54 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project

from types import SimpleNamespace

import pytest

from vllm.model_executor.models.deepseek_v2 import (
_skip_disabled_mtp_weight,
get_spec_layer_idx_from_weight_name,
)


def test_disabled_mtp_checkpoint_weights_are_skipped():
config = SimpleNamespace(num_hidden_layers=78, num_nextn_predict_layers=0)

assert not _skip_disabled_mtp_weight(config, "model.layers.77.self_attn.weight")
assert _skip_disabled_mtp_weight(config, "model.layers.78.self_attn.weight")


def test_spec_layer_index_uses_valid_layer_counts():
config = SimpleNamespace(num_hidden_layers=78, num_nextn_predict_layers=3)

assert get_spec_layer_idx_from_weight_name(config, "model.layers.79.weight") == 79
assert get_spec_layer_idx_from_weight_name(config, "layers.80.weight") == 80
assert get_spec_layer_idx_from_weight_name(config, "model.layers.81.weight") is None


def test_spec_layer_index_does_not_scan_configured_layer_count():
config = SimpleNamespace(
num_hidden_layers=78,
num_nextn_predict_layers=10**12,
)

assert get_spec_layer_idx_from_weight_name(config, "model.layers.80.weight") == 80
assert (
get_spec_layer_idx_from_weight_name(config, "model.embed_tokens.weight") is None
)


@pytest.mark.parametrize("invalid", [None, True, "3", 3.0, -1, [3]])
def test_malformed_mtp_layer_counts_fail_closed(invalid):
valid = SimpleNamespace(num_hidden_layers=78, num_nextn_predict_layers=3)
bad_nextn = SimpleNamespace(num_hidden_layers=78, num_nextn_predict_layers=invalid)
bad_hidden = SimpleNamespace(num_hidden_layers=invalid, num_nextn_predict_layers=3)

for config in (bad_nextn, bad_hidden):
assert not _skip_disabled_mtp_weight(config, "model.layers.78.weight")
assert (
get_spec_layer_idx_from_weight_name(config, "model.layers.78.weight")
is None
)

assert get_spec_layer_idx_from_weight_name(valid, "model.layers.78.weight") == 78
Loading
Loading