From 53a587a18c1061f07b4dd9c1a982771802e63a72 Mon Sep 17 00:00:00 2001 From: pichangping <1337510399@qq.com> Date: Mon, 23 Mar 2026 21:47:05 +0800 Subject: [PATCH 1/3] bugfix: Fixed the error issue when overlaying MTP and full decode on DSV3.1 C8. Signed-off-by: pichangping <1337510399@qq.com> --- vllm_ascend/attention/mla_v1.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/vllm_ascend/attention/mla_v1.py b/vllm_ascend/attention/mla_v1.py index aa93d460572f..b52462f580d1 100644 --- a/vllm_ascend/attention/mla_v1.py +++ b/vllm_ascend/attention/mla_v1.py @@ -1317,6 +1317,8 @@ def _forward_decode( sparse_mode = 3 attn_mask = attn_metadata.decode.attn_mask # type:ignore actual_seq_lengths = decode_meta.actual_seq_lengths_q + if self.fa_quant_layer: + dequant_scale_q_nope = dequant_scale_q_nope.view(num_tokens, self.num_heads) elif self.fa_quant_layer: attn_mask = None input_layout = "BSND_NBSD" @@ -1402,7 +1404,8 @@ def _forward_decode( weak_ref_tensors(softmax_lse), ) if self.fa_quant_layer: - attn_params = attn_params + (dequant_scale_q_nope, self.fak_descale_float) # type: ignore + attn_params = attn_params + ( + weak_ref_tensors(dequant_scale_q_nope), weak_ref_tensors(self.fak_descale_float)) # type: ignore else: attn_params = attn_params + (None, None) # type: ignore From d5a1ac2cebadbfa732267b86e75f15c494242924 Mon Sep 17 00:00:00 2001 From: pichangping <1337510399@qq.com> Date: Mon, 23 Mar 2026 21:58:03 +0800 Subject: [PATCH 2/3] fix CI Signed-off-by: pichangping <1337510399@qq.com> --- vllm_ascend/attention/mla_v1.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/vllm_ascend/attention/mla_v1.py b/vllm_ascend/attention/mla_v1.py index b52462f580d1..3a1790e8e838 100644 --- a/vllm_ascend/attention/mla_v1.py +++ b/vllm_ascend/attention/mla_v1.py @@ -1405,7 +1405,9 @@ def _forward_decode( ) if self.fa_quant_layer: attn_params = attn_params + ( - weak_ref_tensors(dequant_scale_q_nope), weak_ref_tensors(self.fak_descale_float)) # type: ignore + weak_ref_tensors(dequant_scale_q_nope), + weak_ref_tensors(self.fak_descale_float) + ) # type: ignore else: attn_params = attn_params + (None, None) # type: ignore From fa9d361cdc8fe78ead1f1620d25dc46246f7116e Mon Sep 17 00:00:00 2001 From: pichangping <1337510399@qq.com> Date: Mon, 23 Mar 2026 22:01:05 +0800 Subject: [PATCH 3/3] fix CI Signed-off-by: pichangping <1337510399@qq.com> --- vllm_ascend/attention/mla_v1.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/vllm_ascend/attention/mla_v1.py b/vllm_ascend/attention/mla_v1.py index 3a1790e8e838..626780cdc186 100644 --- a/vllm_ascend/attention/mla_v1.py +++ b/vllm_ascend/attention/mla_v1.py @@ -1406,7 +1406,7 @@ def _forward_decode( if self.fa_quant_layer: attn_params = attn_params + ( weak_ref_tensors(dequant_scale_q_nope), - weak_ref_tensors(self.fak_descale_float) + weak_ref_tensors(self.fak_descale_float), ) # type: ignore else: attn_params = attn_params + (None, None) # type: ignore