Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/vllm-main-verified.commit
Original file line number Diff line number Diff line change
@@ -1 +1 @@
85c09e9885e346ea1612da30ebff5a75f67d2350
54503ecec0f3ac31e5ecfc5f28652e4cc42307b5
2 changes: 1 addition & 1 deletion tests/e2e/conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -1924,7 +1924,7 @@ def qwen_prompt(questions: list[str]) -> list[str]:


def hunyuan_prompt(questions: list[str]) -> list[str]:
placeholder = "<|hy_place▁holder▁no▁100|><|hy_place▁holder▁no▁102|><|hy_place▁holder▁no▁101|>" # noqa: E501
placeholder = "<|hy_place▁holder▁no▁102|>" # noqa: E501
return [f"<|hy_begin▁of▁sentence|>{placeholder}{question}<|hy_User|>" for question in questions]


Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -1137,7 +1137,8 @@ def _get_group_kv_caches(self, group_idx: int, layer_indices: list[int] | None =
if layer_indices is None:
_, layer_indices = self.kv_group2layeridx[group_idx]
layer_index_set = set(layer_indices)
num_attn_module = 2 if self.vllm_config.model_config.hf_text_config.model_type == "longcat_flash" else 1
model_type = self.vllm_config.model_config.hf_text_config.model_type
num_attn_module = 2 if model_type in ("longcat_flash", "longcat_flash_ngram") else 1
from vllm.v1.worker.utils import extract_layer_index

def layer_in_group(layer_name: str) -> bool:
Expand Down Expand Up @@ -2125,7 +2126,8 @@ def _build_kv_group2layeridx(self) -> dict[int, tuple[dict[str, Any], list[int]]
from vllm.v1.worker.utils import extract_layer_index

kv_group2layeridx: dict[int, tuple[dict[str, Any], list[int]]] = {}
num_attn_module = 2 if self.vllm_config.model_config.hf_text_config.model_type == "longcat_flash" else 1
model_type = self.vllm_config.model_config.hf_text_config.model_type
num_attn_module = 2 if model_type in ("longcat_flash", "longcat_flash_ngram") else 1
next_mtp_layer_idx = self.total_layers
transfer_group_id = 0
for kv_cache_group_id, group_spec in enumerate(self.kv_cache_config.kv_cache_groups):
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -1336,7 +1336,8 @@ def register_kv_caches(self, kv_caches: dict[str, torch.Tensor]):
if use_kv_buffer:
self.create_kv_buffer(kv_buffer)

num_attn_module = 2 if self.vllm_config.model_config.hf_text_config.model_type == "longcat_flash" else 1
model_type = self.vllm_config.model_config.hf_text_config.model_type
num_attn_module = 2 if model_type in ("longcat_flash", "longcat_flash_ngram") else 1
mtp_layer_name = ""
for layer_name in kv_caches:
if "mtp" in layer_name:
Expand Down
3 changes: 1 addition & 2 deletions vllm_ascend/patch/hunyuan_vl_processor_compat.py
Original file line number Diff line number Diff line change
Expand Up @@ -148,8 +148,7 @@ def call_hf_processor(

def install_hunyuan_vl_processor_compat() -> None:
"""Align both supported vLLM refs with Transformers 5.13 Hunyuan APIs."""
if not _remove_stale_registry_entries():
return
_remove_stale_registry_entries()
from vllm.model_executor.models import hunyuan_vision as main_hunyuan_vision

_patch_hunyuan_processor_loader(main_hunyuan_vision)
Expand Down
2 changes: 1 addition & 1 deletion vllm_ascend/patch/platform/patch_speculative_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -111,7 +111,7 @@ def hf_config_override(hf_config: PretrainedConfig) -> PretrainedConfig:
"architectures": ["Qwen3_5MoeMTP" if is_moe else "Qwen3_5MTP"],
}
)
if hf_config.model_type == "longcat_flash":
if hf_config.model_type in ("longcat_flash", "longcat_flash_ngram"):
hf_config.model_type = "longcat_flash_mtp"
n_predict = getattr(hf_config, "num_nextn_predict_layers", 1)
hf_config.update({"n_predict": n_predict, "architectures": ["LongCatFlashMTPModel"]})
Expand Down
3 changes: 2 additions & 1 deletion vllm_ascend/worker/model_runner_v1.py
Original file line number Diff line number Diff line change
Expand Up @@ -3971,7 +3971,8 @@ def initialize_kv_cache_tensors(self, kv_cache_config: KVCacheConfig) -> dict[st
else:
from vllm.v1.worker.utils import bind_kv_cache

num_attn_module = 2 if self.model_config.hf_text_config.model_type == "longcat_flash" else 1
model_type = self.model_config.hf_text_config.model_type
num_attn_module = 2 if model_type in ("longcat_flash", "longcat_flash_ngram") else 1
bind_kv_cache(
kv_caches,
self.compilation_config.static_forward_context,
Expand Down
Loading