From 6ec13694c34b1b9905323768bf96c32a3772843c Mon Sep 17 00:00:00 2001 From: Andrii Skliar Date: Thu, 30 Jul 2026 13:12:00 +0200 Subject: [PATCH 1/2] Add quantization configuration support to DSparkMarkovHead and Qwen3DSparkModel Signed-off-by: Andrii Skliar --- vllm/model_executor/models/qwen3_dspark.py | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/vllm/model_executor/models/qwen3_dspark.py b/vllm/model_executor/models/qwen3_dspark.py index aa2a9395de19..d168709fe5e6 100644 --- a/vllm/model_executor/models/qwen3_dspark.py +++ b/vllm/model_executor/models/qwen3_dspark.py @@ -22,6 +22,7 @@ from vllm.config import VllmConfig from vllm.logger import init_logger from vllm.model_executor.layers.logits_processor import LogitsProcessor +from vllm.model_executor.layers.quantization import QuantizationConfig from vllm.model_executor.layers.vocab_parallel_embedding import ( ParallelLMHead, ) @@ -51,13 +52,20 @@ def __init__( draft_vocab_size: int, markov_rank: int, prefix: str, + quant_config: QuantizationConfig | None = None, ) -> None: super().__init__() self.markov_w1 = nn.Embedding(vocab_size, markov_rank) + # markov_w2 carries the drafter quant_config so W4A16 heads (which ship + # a quantized markov_w2 with weight_scale_2) load correctly. When the + # prefix is not in the quantized_layers map (unquantized drafter, or a + # mixed-precision config that leaves markov_w2 in bf16) the quant method + # resolves to an unquantized linear, so this is a no-op in those cases. self.markov_w2 = ParallelLMHead( draft_vocab_size, markov_rank, bias=False, + quant_config=quant_config, prefix=maybe_prefix(prefix, "markov_w2"), disable_tp=True, ) @@ -97,6 +105,7 @@ def __init__( draft_vocab_size, config.markov_rank, prefix=maybe_prefix(prefix, "markov_head"), + quant_config=self.quant_config, ) From 5dbdce4b9dd6f066e0f3ef68051a31f8ff0c6e2c Mon Sep 17 00:00:00 2001 From: Andrii Skliar Date: Thu, 30 Jul 2026 13:21:27 +0200 Subject: [PATCH 2/2] remove unnecessary commits Signed-off-by: Andrii Skliar --- vllm/model_executor/models/qwen3_dspark.py | 5 ----- 1 file changed, 5 deletions(-) diff --git a/vllm/model_executor/models/qwen3_dspark.py b/vllm/model_executor/models/qwen3_dspark.py index d168709fe5e6..04af45669498 100644 --- a/vllm/model_executor/models/qwen3_dspark.py +++ b/vllm/model_executor/models/qwen3_dspark.py @@ -56,11 +56,6 @@ def __init__( ) -> None: super().__init__() self.markov_w1 = nn.Embedding(vocab_size, markov_rank) - # markov_w2 carries the drafter quant_config so W4A16 heads (which ship - # a quantized markov_w2 with weight_scale_2) load correctly. When the - # prefix is not in the quantized_layers map (unquantized drafter, or a - # mixed-precision config that leaves markov_w2 in bf16) the quant method - # resolves to an unquantized linear, so this is a no-op in those cases. self.markov_w2 = ParallelLMHead( draft_vocab_size, markov_rank,