Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions .github/workflows/configs/nightly_config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -237,6 +237,10 @@ a3:
multi_card:
test_config:
# pytest-driven tests
- name: kimi-k3-execution-parity
os: linux-aarch64-nightly-a3-16
tests: tests/e2e/nightly/single_node/models/test_kimi_k3_execution_parity.py
testcase_timeout: 180
- name: qwen3-30b-acc
os: linux-aarch64-nightly-a3-4
tests: tests/e2e/weekly/single_node/models/test_qwen3_30b_acc.py
Expand Down
1 change: 1 addition & 0 deletions .github/workflows/scripts/test_config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -307,6 +307,7 @@
- tests/ut/models/test_deepseek_v4_compressor.py
- tests/ut/models/test_deepseek_v4_indexer.py
- tests/ut/models/test_deepseek_v4_moe.py
- tests/ut/models/test_kimi_k3_adapter.py
- tests/e2e/pull_request/four_card/test_deepseek_v4.py

- name: models_minimax_m3
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,63 @@
{
"activation_situ_beta": 4,
"activation_situ_linear_beta": 25,
"architectures": [
"KimiLinearForCausalLM"
],
"attn_res_block_size": 12,
"bos_token_id": 163584,
"dtype": "bfloat16",
"eos_token_id": 163586,
"first_k_dense_replace": 1,
"hidden_act": "situ",
"hidden_size": 7168,
"intermediate_size": 33792,
"kv_lora_rank": 512,
"latent_moe_use_norm": true,
"linear_attn_config": {
"full_attn_layers": [
4
],
"gate_lower_bound": -5,
"head_dim": 128,
"kda_layers": [
1,
2,
3,
5
],
"num_heads": 96,
"short_conv_kernel_size": 4,
"use_full_rank_gate": true
},
"max_position_embeddings": 1048576,
"mla_use_nope": true,
"mla_use_output_gate": true,
"model_type": "kimi_linear",
"moe_intermediate_size": 3072,
"moe_layer_freq": 1,
"moe_renormalize": true,
"moe_router_activation_func": "sigmoid",
"num_attention_heads": 96,
"num_expert_group": 1,
"num_experts": 16,
"num_experts_per_token": 16,
"num_hidden_layers": 5,
"num_key_value_heads": 96,
"num_nextn_predict_layers": 0,
"num_shared_experts": 2,
"pad_token_id": 163839,
"q_lora_rank": 1536,
"qk_nope_head_dim": 128,
"qk_rope_head_dim": 64,
"rms_norm_eps": 1e-05,
"routed_expert_hidden_size": 3584,
"routed_scaling_factor": 1,
"tie_word_embeddings": false,
"topk_group": 1,
"topk_method": "noaux_tc",
"use_cache": true,
"use_grouped_topk": true,
"v_head_dim": 128,
"vocab_size": 163840
}
110 changes: 110 additions & 0 deletions tests/e2e/nightly/single_node/models/test_kimi_k3_execution_parity.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,110 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
"""Storage-light Kimi K3 execution parity guard.

The committed fixture keeps Kimi K3's production dimensions and its mixed
KDA/MLA layout, but limits the model to five layers and sixteen experts. Dummy
weights deliberately make this an execution-parity test, not a semantic
accuracy test. Full-checkpoint GPQA remains a separate release gate.
"""

from pathlib import Path

import pytest
import torch
from vllm import SamplingParams
from vllm.inputs import TokensPrompt

from tests.e2e.conftest import VllmRunner, wait_until_npu_memory_free

MODEL_CONFIG = Path(__file__).parent / "fixtures" / "kimi_k3_5layers_16experts"
SCHEDULER_BLOCK_SIZE = 16
PROMPT_TOKEN_IDS = [163584, *range(100, 100 + SCHEDULER_BLOCK_SIZE)]
MAX_TOKENS = 4


def _assert_complete_output(request_output):
assert request_output is not None
assert request_output.finished
assert request_output.outputs is not None
assert len(request_output.outputs) == 1

completion = request_output.outputs[0]
assert completion is not None
assert completion.token_ids is not None
assert len(completion.token_ids) == MAX_TOKENS
assert completion.logprobs is not None
assert len(completion.logprobs) == MAX_TOKENS

chosen_logprobs = []
for token_id, step_logprobs in zip(completion.token_ids, completion.logprobs):
assert step_logprobs is not None
assert token_id in step_logprobs
logprob = step_logprobs[token_id].logprob
assert logprob is not None
assert torch.isfinite(torch.tensor(logprob))
chosen_logprobs.append(logprob)

return list(completion.token_ids), torch.tensor(chosen_logprobs, dtype=torch.float32)


@pytest.mark.e2e_model("sgl-npu/Kimi-K3-W4A8")
@pytest.mark.e2e_coverage(
arch="moe",
feature="aclgraph,prefix_caching,logprobs",
parallel="TP,EP",
deploy="pd_mix",
hardware="A3",
quantization="BF16",
graph_mode="full_decode_only",
)
@wait_until_npu_memory_free()
def test_kimi_k3_dummy_prefix_cache_one_token_prefill_parity():
"""Compare cold prefill with the cached block-size-plus-one path."""
sampling_params = SamplingParams(
temperature=0,
max_tokens=MAX_TOKENS,
logprobs=1,
ignore_eos=True,
seed=0,
)
prompt = TokensPrompt(prompt_token_ids=PROMPT_TOKEN_IDS)

with VllmRunner(
str(MODEL_CONFIG),
skip_tokenizer_init=True,
load_format="dummy",
dtype="bfloat16",
seed=0,
block_size=SCHEDULER_BLOCK_SIZE,
max_model_len=64,
max_num_seqs=1,
max_num_batched_tokens=64,
tensor_parallel_size=16,
enable_expert_parallel=True,
enable_prefix_caching=True,
gpu_memory_utilization=0.75,
compilation_config={
"cudagraph_mode": "FULL_DECODE_ONLY",
"cudagraph_capture_sizes": [1],
},
) as vllm_model:
cold = vllm_model.model.generate([prompt], sampling_params, use_tqdm=False)[0]
hit = vllm_model.model.generate([prompt], sampling_params, use_tqdm=False)[0]

assert cold.num_cached_tokens in (None, 0)
assert hit.num_cached_tokens == SCHEDULER_BLOCK_SIZE
assert len(PROMPT_TOKEN_IDS) - hit.num_cached_tokens == 1

cold_tokens, cold_logprobs = _assert_complete_output(cold)
hit_tokens, hit_logprobs = _assert_complete_output(hit)
assert hit_tokens == cold_tokens
torch.testing.assert_close(hit_logprobs, cold_logprobs, rtol=5e-3, atol=5e-3)

assert vllm_model.model.reset_prefix_cache()
reset = vllm_model.model.generate([prompt], sampling_params, use_tqdm=False)[0]
assert reset.num_cached_tokens in (None, 0)
reset_tokens, reset_logprobs = _assert_complete_output(reset)
assert reset_tokens == cold_tokens
torch.testing.assert_close(reset_logprobs, cold_logprobs, rtol=5e-3, atol=5e-3)
94 changes: 93 additions & 1 deletion tests/ut/model_executor/test_qwen3_dspark.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,11 +19,16 @@

from __future__ import annotations

import json
from types import SimpleNamespace
from unittest.mock import patch

import torch
from safetensors.torch import save_file
from torch import nn

import vllm_ascend.models.qwen3_dspark as qwen3_dspark
from vllm_ascend.models.llama_eagle3 import load_quarot_target_layer


class TestQwen3DSparkWeightLoading:
Expand All @@ -45,7 +50,11 @@ def test_rotates_only_fc_weights(self) -> None:
rotation_matrix = torch.tensor([[0.0, 1.0], [1.0, 0.0]])
fc_weight = torch.tensor([[1.0, 2.0], [3.0, 4.0]])
non_fc_weight = torch.tensor([[5.0, 6.0]])
weights_to_load = [("model.fc.weight", fc_weight), ("model.embed_tokens.weight", non_fc_weight)]
weights_to_load = [
("model.fc.weight", fc_weight),
("model.embed_tokens.weight", non_fc_weight),
("lm_head.weight", non_fc_weight),
]
expected_fc_weight = torch.matmul(fc_weight, rotation_matrix)

# Capture the final delegation without invoking the real model loader.
Expand All @@ -63,3 +72,86 @@ def test_rotates_only_fc_weights(self) -> None:
processed_weights = mock_parent_load_weights.call_args.args[0]
torch.testing.assert_close(processed_weights[0][1], expected_fc_weight)
torch.testing.assert_close(processed_weights[1][1], non_fc_weight)
torch.testing.assert_close(processed_weights[2][1], non_fc_weight)

def test_quarot_loads_missing_boundaries_in_modeling(self) -> None:
model_cls = qwen3_dspark.AscendQwen3DSparkForCausalLM
model = model_cls.__new__(model_cls)
nn.Module.__init__(model)
model.rotation_path = "quarot.safetensors"
model.target_model_path = "/target"
model.enable_confidence_head = False
model.model = SimpleNamespace(embed_tokens=object())
model.lm_head = object()
rotation = torch.eye(2)

with (
patch.object(
qwen3_dspark,
"get_rotation_matrix",
return_value=rotation,
),
patch.object(
qwen3_dspark,
"load_quarot_target_layer",
) as load_target_layer,
patch.object(
qwen3_dspark.Qwen3DSparkForCausalLM,
"load_weights",
),
):
model.load_weights([("model.fc.weight", torch.eye(2))])

assert load_target_layer.call_count == 2
assert load_target_layer.call_args_list[0].args[:2] == (
model.model.embed_tokens,
model.target_model_path,
)
assert load_target_layer.call_args_list[1].args[:2] == (
model.lm_head,
model.target_model_path,
)
assert model.has_own_embed_tokens
assert model.has_own_lm_head


def test_load_quarot_target_layer_reads_local_vocab_shard(tmp_path) -> None:
weight_name = "language_model.model.embed_tokens.weight"
shard_name = "model-00001-of-00001.safetensors"
target_weight = torch.tensor(
[
[1.0, 2.0],
[3.0, 4.0],
[5.0, 6.0],
[7.0, 8.0],
]
)
save_file({weight_name: target_weight}, tmp_path / shard_name)
(tmp_path / "model.safetensors.index.json").write_text(
json.dumps({"weight_map": {weight_name: shard_name}}),
encoding="utf-8",
)

layer = nn.Linear(2, 3, bias=False)
layer.weight.data.fill_(99)
layer.shard_indices = SimpleNamespace(
org_vocab_start_index=1,
org_vocab_end_index=3,
)
rotation = torch.tensor([[0.0, 1.0], [1.0, 0.0]])

load_quarot_target_layer(
layer,
tmp_path,
(weight_name,),
rotation,
"test embedding",
)

expected = torch.cat(
(
target_weight[1:3] @ rotation.T,
torch.zeros(1, 2),
)
)
torch.testing.assert_close(layer.weight, expected)
Loading
Loading