diff --git a/.github/vllm-main-verified.commit b/.github/vllm-main-verified.commit index 6be422f3d04b..3bc0c33c8aaa 100644 --- a/.github/vllm-main-verified.commit +++ b/.github/vllm-main-verified.commit @@ -1 +1 @@ -85c09e9885e346ea1612da30ebff5a75f67d2350 +54503ecec0f3ac31e5ecfc5f28652e4cc42307b5 diff --git a/tests/e2e/conftest.py b/tests/e2e/conftest.py index f58e584b9cfe..9ec9d5f80eac 100644 --- a/tests/e2e/conftest.py +++ b/tests/e2e/conftest.py @@ -1924,7 +1924,7 @@ def qwen_prompt(questions: list[str]) -> list[str]: def hunyuan_prompt(questions: list[str]) -> list[str]: - placeholder = "<|hy_place▁holder▁no▁100|><|hy_place▁holder▁no▁102|><|hy_place▁holder▁no▁101|>" # noqa: E501 + placeholder = "<|hy_place▁holder▁no▁102|>" # noqa: E501 return [f"<|hy_begin▁of▁sentence|>{placeholder}{question}<|hy_User|>" for question in questions] diff --git a/vllm_ascend/distributed/kv_transfer/kv_p2p/mooncake_connector.py b/vllm_ascend/distributed/kv_transfer/kv_p2p/mooncake_connector.py index 669031d34665..5101fb18d326 100644 --- a/vllm_ascend/distributed/kv_transfer/kv_p2p/mooncake_connector.py +++ b/vllm_ascend/distributed/kv_transfer/kv_p2p/mooncake_connector.py @@ -1137,7 +1137,8 @@ def _get_group_kv_caches(self, group_idx: int, layer_indices: list[int] | None = if layer_indices is None: _, layer_indices = self.kv_group2layeridx[group_idx] layer_index_set = set(layer_indices) - num_attn_module = 2 if self.vllm_config.model_config.hf_text_config.model_type == "longcat_flash" else 1 + model_type = self.vllm_config.model_config.hf_text_config.model_type + num_attn_module = 2 if model_type in ("longcat_flash", "longcat_flash_ngram") else 1 from vllm.v1.worker.utils import extract_layer_index def layer_in_group(layer_name: str) -> bool: @@ -2125,7 +2126,8 @@ def _build_kv_group2layeridx(self) -> dict[int, tuple[dict[str, Any], list[int]] from vllm.v1.worker.utils import extract_layer_index kv_group2layeridx: dict[int, tuple[dict[str, Any], list[int]]] = {} - num_attn_module = 2 if self.vllm_config.model_config.hf_text_config.model_type == "longcat_flash" else 1 + model_type = self.vllm_config.model_config.hf_text_config.model_type + num_attn_module = 2 if model_type in ("longcat_flash", "longcat_flash_ngram") else 1 next_mtp_layer_idx = self.total_layers transfer_group_id = 0 for kv_cache_group_id, group_spec in enumerate(self.kv_cache_config.kv_cache_groups): diff --git a/vllm_ascend/distributed/kv_transfer/kv_p2p/mooncake_layerwise_connector.py b/vllm_ascend/distributed/kv_transfer/kv_p2p/mooncake_layerwise_connector.py index 2d779994dc37..5d844aa69dc1 100644 --- a/vllm_ascend/distributed/kv_transfer/kv_p2p/mooncake_layerwise_connector.py +++ b/vllm_ascend/distributed/kv_transfer/kv_p2p/mooncake_layerwise_connector.py @@ -1336,7 +1336,8 @@ def register_kv_caches(self, kv_caches: dict[str, torch.Tensor]): if use_kv_buffer: self.create_kv_buffer(kv_buffer) - num_attn_module = 2 if self.vllm_config.model_config.hf_text_config.model_type == "longcat_flash" else 1 + model_type = self.vllm_config.model_config.hf_text_config.model_type + num_attn_module = 2 if model_type in ("longcat_flash", "longcat_flash_ngram") else 1 mtp_layer_name = "" for layer_name in kv_caches: if "mtp" in layer_name: diff --git a/vllm_ascend/patch/hunyuan_vl_processor_compat.py b/vllm_ascend/patch/hunyuan_vl_processor_compat.py index 9ba86091177d..370d1ff86f69 100644 --- a/vllm_ascend/patch/hunyuan_vl_processor_compat.py +++ b/vllm_ascend/patch/hunyuan_vl_processor_compat.py @@ -148,8 +148,7 @@ def call_hf_processor( def install_hunyuan_vl_processor_compat() -> None: """Align both supported vLLM refs with Transformers 5.13 Hunyuan APIs.""" - if not _remove_stale_registry_entries(): - return + _remove_stale_registry_entries() from vllm.model_executor.models import hunyuan_vision as main_hunyuan_vision _patch_hunyuan_processor_loader(main_hunyuan_vision) diff --git a/vllm_ascend/patch/platform/patch_speculative_config.py b/vllm_ascend/patch/platform/patch_speculative_config.py index c27986884ccd..7ba0eaa0f472 100644 --- a/vllm_ascend/patch/platform/patch_speculative_config.py +++ b/vllm_ascend/patch/platform/patch_speculative_config.py @@ -111,7 +111,7 @@ def hf_config_override(hf_config: PretrainedConfig) -> PretrainedConfig: "architectures": ["Qwen3_5MoeMTP" if is_moe else "Qwen3_5MTP"], } ) - if hf_config.model_type == "longcat_flash": + if hf_config.model_type in ("longcat_flash", "longcat_flash_ngram"): hf_config.model_type = "longcat_flash_mtp" n_predict = getattr(hf_config, "num_nextn_predict_layers", 1) hf_config.update({"n_predict": n_predict, "architectures": ["LongCatFlashMTPModel"]}) diff --git a/vllm_ascend/worker/model_runner_v1.py b/vllm_ascend/worker/model_runner_v1.py index 6116aed01217..51dabae6c432 100644 --- a/vllm_ascend/worker/model_runner_v1.py +++ b/vllm_ascend/worker/model_runner_v1.py @@ -3971,7 +3971,8 @@ def initialize_kv_cache_tensors(self, kv_cache_config: KVCacheConfig) -> dict[st else: from vllm.v1.worker.utils import bind_kv_cache - num_attn_module = 2 if self.model_config.hf_text_config.model_type == "longcat_flash" else 1 + model_type = self.model_config.hf_text_config.model_type + num_attn_module = 2 if model_type in ("longcat_flash", "longcat_flash_ngram") else 1 bind_kv_cache( kv_caches, self.compilation_config.static_forward_context,