From 0d81ab21cf2e0886234a88a47d14c6131dd95681 Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Thu, 10 Sep 2026 02:43:18 +0000 Subject: [PATCH 01/17] feat(platform): enable model runner v2 by default via whitelists Rebase https://github.com/vllm-project/vllm-ascend/pull/11692 onto current vllm-project/vllm-ascend main. Replace the env-only use_v2_model_runner override with Ascend-owned whitelist heuristics (Qwen3ForCausalLM; eagle3/mtp/dflash; Triton; non-310P). Keep later unsupported-feature patches for spec-PP and Ascend-supported V1 features (dspark/dflash2). Explicit VLLM_USE_V2_MODEL_RUNNER still wins when set. Signed-off-by: yjyang62 --- .../one_card/test_attention_v1_precision.py | 8 +- .../test_patch_use_v2_model_runner.py | 32 ++- tests/ut/test_mrv2_utils.py | 219 ++++++++++++++++++ vllm_ascend/__init__.py | 2 + vllm_ascend/mrv2_utils.py | 159 +++++++++++++ vllm_ascend/patch/__init__.py | 17 +- .../platform/patch_use_v2_model_runner.py | 41 +--- vllm_ascend/patch/worker/__init__.py | 7 +- vllm_ascend/platform.py | 13 ++ vllm_ascend/worker/worker.py | 7 +- 10 files changed, 459 insertions(+), 46 deletions(-) create mode 100644 tests/ut/test_mrv2_utils.py create mode 100644 vllm_ascend/mrv2_utils.py diff --git a/tests/e2e/pull_request/one_card/test_attention_v1_precision.py b/tests/e2e/pull_request/one_card/test_attention_v1_precision.py index 4bd3e7980c76..a9e72cc3a467 100644 --- a/tests/e2e/pull_request/one_card/test_attention_v1_precision.py +++ b/tests/e2e/pull_request/one_card/test_attention_v1_precision.py @@ -21,7 +21,13 @@ @pytest.fixture(autouse=True) -def default_vllm_config(): +def default_vllm_config(monkeypatch): + # Qwen3 now defaults to the V2 model runner, whose forward-context hook + # (NPUPlatform.set_additional_forward_context) queries TP/DP groups. This + # suite drives attention backends in a single process with no engine, so + # mock the group accessors, mirroring patch_distributed_groups above. + monkeypatch.setattr("vllm.distributed.get_tensor_model_parallel_world_size", lambda: 1) + monkeypatch.setattr("vllm.distributed.get_dp_group", lambda: MagicMock(world_size=1)) mock_config = MagicMock() mock_config.compilation_config = MagicMock() mock_config.compilation_config.custom_ops = ["all"] diff --git a/tests/ut/patch/platform/test_patch_use_v2_model_runner.py b/tests/ut/patch/platform/test_patch_use_v2_model_runner.py index 2a83ee35273e..5ff8243a554f 100644 --- a/tests/ut/patch/platform/test_patch_use_v2_model_runner.py +++ b/tests/ut/patch/platform/test_patch_use_v2_model_runner.py @@ -1,13 +1,19 @@ +import pytest from vllm.config.vllm import VllmConfig from vllm_ascend.patch.platform import patch_use_v2_model_runner -from vllm_ascend.utils import vllm_version_is + + +def test_use_v2_model_runner_is_driven_by_ascend_whitelist(): + assert isinstance(VllmConfig.use_v2_model_runner, property) + from vllm_ascend.mrv2_utils import use_v2_model_runner + + assert VllmConfig.use_v2_model_runner.fget is use_v2_model_runner def test_ascend_v1_supported_features_are_not_rejected(monkeypatch): - if vllm_version_is("0.28.0"): - assert "_get_v1_model_runner_unsupported_features" not in VllmConfig.__dict__ - return + if not hasattr(VllmConfig, "_get_v1_model_runner_unsupported_features"): + pytest.skip("V1 model runner validation is only present on vLLM main") monkeypatch.setattr( patch_use_v2_model_runner, @@ -23,3 +29,21 @@ def test_ascend_v1_supported_features_are_not_rejected(monkeypatch): unsupported = patch_use_v2_model_runner._patched_get_v1_model_runner_unsupported_features(object()) assert unsupported == ["prefill context parallel", "diffusion models"] + + +def test_release_pcp_is_not_rejected_as_v2_unsupported_feature(monkeypatch): + monkeypatch.setattr(patch_use_v2_model_runner, "vllm_version_is", lambda version: version == "0.28.0") + monkeypatch.setattr( + patch_use_v2_model_runner, + "_original_get_unsupported_features", + lambda _: ["prefill context parallelism", "diffusion models"], + ) + monkeypatch.setattr( + patch_use_v2_model_runner, + "resolve_spec_pp_support", + lambda _: None, + ) + + unsupported = patch_use_v2_model_runner._patched_get_unsupported_features(object()) + + assert unsupported == ["diffusion models"] diff --git a/tests/ut/test_mrv2_utils.py b/tests/ut/test_mrv2_utils.py new file mode 100644 index 000000000000..7429185bd1ca --- /dev/null +++ b/tests/ut/test_mrv2_utils.py @@ -0,0 +1,219 @@ +# +# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# This file is a part of the vllm-ascend project. +# + +from types import SimpleNamespace + +import pytest + +import vllm_ascend.mrv2_utils as mrv2_utils +from vllm_ascend.mrv2_utils import ( + _v2_model_runner_environment_ready, + is_default_v2_model_runner_model, + is_supported_v2_model_runner_feature, + use_v2_model_runner, +) + +DEFAULT_V2_ARCH = "Qwen3ForCausalLM" + + +def _make_model_config(**kwargs) -> SimpleNamespace: + attrs = { + "runner_type": "generate", + "is_hybrid": False, + "is_attention_free": False, + "architectures": ["SomeModelForCausalLM"], + } + attrs.update(kwargs) + return SimpleNamespace(**attrs) + + +def _make_vllm_config( + model_config=None, + speculative_config=None, +) -> SimpleNamespace: + return SimpleNamespace( + model_config=model_config, + speculative_config=speculative_config, + ) + + +class TestIsDefaultV2ModelRunnerModel: + def test_whitelisted_architecture(self): + config = _make_vllm_config(model_config=_make_model_config(architectures=[DEFAULT_V2_ARCH])) + + assert is_default_v2_model_runner_model(config) is True + + def test_unknown_architecture(self): + config = _make_vllm_config(model_config=_make_model_config()) + + assert is_default_v2_model_runner_model(config) is False + + def test_none_model_config(self): + assert is_default_v2_model_runner_model(_make_vllm_config(model_config=None)) is False + + def test_non_generate_runner_type(self): + config = _make_vllm_config( + model_config=_make_model_config(runner_type="embedding", architectures=[DEFAULT_V2_ARCH]) + ) + + assert is_default_v2_model_runner_model(config) is False + + def test_hybrid_model(self): + config = _make_vllm_config(model_config=_make_model_config(is_hybrid=True, architectures=[DEFAULT_V2_ARCH])) + + assert is_default_v2_model_runner_model(config) is False + + def test_attention_free_model(self): + config = _make_vllm_config( + model_config=_make_model_config(is_attention_free=True, architectures=[DEFAULT_V2_ARCH]) + ) + + assert is_default_v2_model_runner_model(config) is False + + +class TestIsSupportedV2ModelRunnerFeature: + def test_without_speculative_config(self): + config = _make_vllm_config(speculative_config=None) + + assert is_supported_v2_model_runner_feature(config) is True + + @pytest.mark.parametrize("method", ["eagle3", "mtp", "dflash"]) + def test_whitelisted_methods(self, monkeypatch, method): + monkeypatch.setattr(mrv2_utils.logger, "info_once", lambda *args: None) + config = _make_vllm_config(speculative_config=SimpleNamespace(method=method)) + + assert is_supported_v2_model_runner_feature(config) is True + + @pytest.mark.parametrize("method", ["ngram", "ngram_gpu", "eagle", "unknown_method"]) + def test_unsupported_method(self, method): + config = _make_vllm_config(speculative_config=SimpleNamespace(method=method)) + + assert is_supported_v2_model_runner_feature(config) is False + + def test_whitelisted_method_logs_info(self, monkeypatch): + info_calls = [] + monkeypatch.setattr(mrv2_utils.logger, "info_once", lambda *args: info_calls.append(args)) + config = _make_vllm_config(speculative_config=SimpleNamespace(method="eagle3")) + + assert is_supported_v2_model_runner_feature(config) is True + assert len(info_calls) == 1 + + +class TestV2ModelRunnerEnvironmentReady: + def test_unsupported_feature(self): + config = _make_vllm_config(speculative_config=SimpleNamespace(method="ngram")) + + assert _v2_model_runner_environment_ready(config) is False + + def test_without_triton_on_non_310p(self, monkeypatch): + monkeypatch.setattr(mrv2_utils, "is_310p", lambda: False) + monkeypatch.setattr("vllm.triton_utils.HAS_TRITON", False) + warning_calls = [] + monkeypatch.setattr(mrv2_utils.logger, "warning_once", lambda *args: warning_calls.append(args)) + config = _make_vllm_config(speculative_config=None) + + assert _v2_model_runner_environment_ready(config) is False + assert len(warning_calls) == 1 + + def test_with_triton_on_non_310p(self, monkeypatch): + monkeypatch.setattr(mrv2_utils, "is_310p", lambda: False) + monkeypatch.setattr("vllm.triton_utils.HAS_TRITON", True) + config = _make_vllm_config(speculative_config=None) + + assert _v2_model_runner_environment_ready(config) is True + + @pytest.mark.parametrize("has_triton", [True, False]) + def test_310p_excluded_regardless_of_triton(self, monkeypatch, has_triton): + # 310P does not support the V2 model runner. + monkeypatch.setattr(mrv2_utils, "is_310p", lambda: True) + monkeypatch.setattr("vllm.triton_utils.HAS_TRITON", has_triton) + config = _make_vllm_config(speculative_config=None) + + assert _v2_model_runner_environment_ready(config) is False + + +class TestUseV2ModelRunner: + @pytest.mark.parametrize("env_value", [True, False]) + def test_env_override_wins(self, monkeypatch, env_value): + monkeypatch.setattr(mrv2_utils.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", env_value) + config = _make_vllm_config(model_config=_make_model_config()) + + assert use_v2_model_runner(config) is env_value + + def test_default_enabled_for_whitelisted_model(self, monkeypatch): + monkeypatch.setattr(mrv2_utils.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", None) + monkeypatch.setattr(mrv2_utils, "_v2_model_runner_environment_ready", lambda _config: True) + config = _make_vllm_config(model_config=_make_model_config(architectures=[DEFAULT_V2_ARCH])) + + assert use_v2_model_runner(config) is True + + def test_default_disabled_when_environment_not_ready(self, monkeypatch): + monkeypatch.setattr(mrv2_utils.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", None) + monkeypatch.setattr(mrv2_utils, "_v2_model_runner_environment_ready", lambda _config: False) + config = _make_vllm_config(model_config=_make_model_config(architectures=[DEFAULT_V2_ARCH])) + + assert use_v2_model_runner(config) is False + + def test_non_whitelisted_model_falls_back(self, monkeypatch): + monkeypatch.setattr(mrv2_utils.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", None) + warning_calls = [] + monkeypatch.setattr(mrv2_utils.logger, "warning_once", lambda *args: warning_calls.append(args)) + config = _make_vllm_config(model_config=_make_model_config()) + + assert use_v2_model_runner(config) is False + assert len(warning_calls) == 1 + + def test_none_model_config_falls_back(self, monkeypatch): + monkeypatch.setattr(mrv2_utils.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", None) + warning_calls = [] + monkeypatch.setattr(mrv2_utils.logger, "warning_once", lambda *args: warning_calls.append(args)) + config = _make_vllm_config(model_config=None) + + assert use_v2_model_runner(config) is False + assert len(warning_calls) == 1 + + +class TestV2ModelRunnerValidationPatch: + def test_validation_is_decoupled_from_upstream(self): + # The Ascend V2 runner decision is fully owned by use_v2_model_runner, + # so the replacement validation must never raise (e.g. the upstream + # Triton / feature-support checks do not apply). + mrv2_utils._validate_v2_model_runner(object()) + mrv2_utils._validate_v2_model_runner(SimpleNamespace()) + + def test_apply_config_patch_is_wired(self, monkeypatch): + from vllm.config.vllm import VllmConfig + + original_property = VllmConfig.use_v2_model_runner + original_validate = VllmConfig._validate_v2_model_runner + + monkeypatch.setattr("vllm.config.vllm.HAS_TRITON", False) + + mrv2_utils.apply_v2_model_runner_config_patch() + assert isinstance(VllmConfig.use_v2_model_runner, property) + assert VllmConfig.use_v2_model_runner.fget is mrv2_utils.use_v2_model_runner + + # The upstream Triton check must no longer run. + VllmConfig._validate_v2_model_runner(object()) + + # Re-applying is harmless. + mrv2_utils.apply_v2_model_runner_config_patch() + VllmConfig._validate_v2_model_runner(object()) + + # Restore the upstream class state so later tests are unaffected. + monkeypatch.setattr(VllmConfig, "use_v2_model_runner", original_property) + monkeypatch.setattr(VllmConfig, "_validate_v2_model_runner", original_validate) diff --git a/vllm_ascend/__init__.py b/vllm_ascend/__init__.py index 1c913936ef47..5044d9cbaed0 100644 --- a/vllm_ascend/__init__.py +++ b/vllm_ascend/__init__.py @@ -59,8 +59,10 @@ def _ensure_global_patch(): if _GLOBAL_PATCH_APPLIED: return + from vllm_ascend.mrv2_utils import apply_v2_model_runner_config_patch from vllm_ascend.utils import adapt_patch + apply_v2_model_runner_config_patch() adapt_patch(is_global_patch=True) _GLOBAL_PATCH_APPLIED = True diff --git a/vllm_ascend/mrv2_utils.py b/vllm_ascend/mrv2_utils.py new file mode 100644 index 000000000000..46a9caa60401 --- /dev/null +++ b/vllm_ascend/mrv2_utils.py @@ -0,0 +1,159 @@ +# +# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# This file is a part of the vllm-ascend project. +# + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import vllm.envs as envs_vllm +from vllm.logger import logger + +if TYPE_CHECKING: + from vllm.config import VllmConfig +else: + VllmConfig = None + +from vllm_ascend.utils import is_310p + +# Architectures for which Model Runner V2 is enabled by default on Ascend. +DEFAULT_V2_MODEL_RUNNER_ARCHITECTURES = frozenset( + { + "Qwen3ForCausalLM", + } +) + + +def _validate_v2_model_runner(vllm_config: VllmConfig) -> None: + """No-op replacement for the upstream V2 model runner validation. + + Ascend fully owns the V2 model runner enablement decision through the model + / feature whitelists in :func:`use_v2_model_runner`, so the upstream checks + -- Triton availability plus the list of features the *upstream* GPU V2 + runner does not yet support -- are intentionally decoupled. Otherwise a V2 + enablement decision made here (e.g. via an explicit + ``VLLM_USE_V2_MODEL_RUNNER=1``) could fail at config construction with + upstream checks that do not apply to the Ascend runner. + """ + + +def apply_v2_model_runner_config_patch() -> None: + """Apply the Ascend V2 model runner overrides to VllmConfig. + + Installs two overrides on the ``VllmConfig`` class: + + * ``use_v2_model_runner`` is driven by the Ascend whitelist default instead + of the upstream default decision (see :func:`use_v2_model_runner`). + * ``_validate_v2_model_runner`` is neutralized because the upstream checks + describe the upstream GPU runner and do not apply to the Ascend runner. + + Must run wherever ``VllmConfig`` is (re)created or its properties are read + in a separate process -- the frontend during config construction, each + worker process, and the engine-core process (the scheduler reads + ``use_v2_model_runner`` there from a pickled config, so the class-level + patch does not carry over from the frontend). Repeated application is + harmless: it just re-assigns the same overrides. + """ + from vllm.config.vllm import VllmConfig + + VllmConfig.use_v2_model_runner = property(use_v2_model_runner) + VllmConfig._validate_v2_model_runner = _validate_v2_model_runner + + +def is_default_v2_model_runner_model(vllm_config: VllmConfig) -> bool: + """Model whitelist: enable V2 for default-V2 architectures.""" + model_config = vllm_config.model_config + if model_config is None: + return False + + if getattr(model_config, "runner_type", "generate") != "generate": + return False + + if getattr(model_config, "is_hybrid", False): + return False + + if getattr(model_config, "is_attention_free", False): + return False + + architectures = getattr(model_config, "architectures", []) + return any(arch in DEFAULT_V2_MODEL_RUNNER_ARCHITECTURES for arch in architectures) + + +def is_supported_v2_model_runner_feature(vllm_config: VllmConfig) -> bool: + """Feature whitelist: only whitelisted features may be enabled with a whitelisted model.""" + speculative_config = vllm_config.speculative_config + if speculative_config is None: + return True + if speculative_config.method in ("eagle3", "mtp", "dflash"): + logger.info_once( + "Model Runner V2 is enabled by default for speculative method '%s'.", + speculative_config.method, + ) + return True + return False + + +def _v2_model_runner_environment_ready(vllm_config: VllmConfig) -> bool: + """Check the remaining V2 gates (feature whitelist + platform + Triton).""" + if not is_supported_v2_model_runner_feature(vllm_config): + return False + + if is_310p(): + logger.warning_once("Model Runner V2 is not supported on 310P; using the V1 model runner instead.") + return False + + from vllm.triton_utils import HAS_TRITON + + if not HAS_TRITON: + logger.warning_once("Model Runner V2 requires Triton; using the V1 model runner instead.") + return False + + return True + + +def use_v2_model_runner(vllm_config: VllmConfig) -> bool: + """Return whether the V2 model runner should be used on Ascend. + + An explicit ``VLLM_USE_V2_MODEL_RUNNER`` override wins. Otherwise the V2 + runner is enabled by default only when all of the following hold: + + * the model is on the default-V2 model whitelist, + * the enabled features are on the V2 feature whitelist, + * the platform is not 310P and the runtime provides Triton. + """ + use_v2_model_runner = envs_vllm.VLLM_USE_V2_MODEL_RUNNER + if use_v2_model_runner is not None: + logger.info_once( + "VLLM_USE_V2_MODEL_RUNNER=%s is set; using Model Runner %s.", + use_v2_model_runner, + "V2" if use_v2_model_runner else "V1", + ) + return use_v2_model_runner + + if is_default_v2_model_runner_model(vllm_config): + if _v2_model_runner_environment_ready(vllm_config): + architectures = getattr(vllm_config.model_config, "architectures", []) + logger.info_once( + "Model Runner V2 is enabled for %s.", + ", ".join(architectures), + ) + return True + return False + + logger.warning_once( + "Model Runner V2 model whitelist does not include this model; using the V1 model runner instead." + ) + return False diff --git a/vllm_ascend/patch/__init__.py b/vllm_ascend/patch/__init__.py index d71d68543756..9fbbca5fbc7e 100644 --- a/vllm_ascend/patch/__init__.py +++ b/vllm_ascend/patch/__init__.py @@ -679,16 +679,20 @@ # architecture whitelists, Triton availability, and feature # compatibility checks. On Ascend the NPU v2 runner is not yet # compatible with all upstream-defaulted models and features, so -# enabling by model architecture can crash. We override the -# property to read only VLLM_USE_V2_MODEL_RUNNER, deferring -# model/framework checks to the NPU runner itself. +# following upstream defaults can crash. We override the property +# with Ascend-owned whitelist heuristics in mrv2_utils (currently +# Qwen3ForCausalLM plus eagle3/mtp/dflash spec decode, Triton, and +# non-310P). Explicit VLLM_USE_V2_MODEL_RUNNER still wins. # How: -# Monkey-patch VllmConfig.use_v2_model_runner to return -# envs.VLLM_USE_V2_MODEL_RUNNER (defaulting to False when unset). +# Call apply_v2_model_runner_config_patch() to install the Ascend +# use_v2_model_runner property and neutralize upstream V2 +# validation. Keep additional patches for V2 spec-PP unsupported +# features and Ascend-supported V1 features (dspark / dflash2). # worker/patch_v2/patch_use_v2_model_runner.py reuses this platform # patch so EngineCore and worker processes share the same behavior. # Related PR (if no, explain why): # 1. https://github.com/vllm-project/vllm-ascend/pull/11389 +# 2. https://github.com/vllm-project/vllm-ascend/pull/11692 # Future Plan: # Remove this patch once vllm-ascend fully supports the v2 model # runner and can rely on upstream's default enablement heuristics @@ -1236,7 +1240,8 @@ # Why: # EngineCore subprocesses only load global/platform patches, while workers # also import this compatibility module. The actual monkey-patch is defined -# in `platform/patch_use_v2_model_runner.py`. +# in `platform/patch_use_v2_model_runner.py` (whitelist default plus +# remaining V2/V1 feature patches). # How: # Reuse the platform patch so EngineCore and worker processes share the # same `use_v2_model_runner` behavior. diff --git a/vllm_ascend/patch/platform/patch_use_v2_model_runner.py b/vllm_ascend/patch/platform/patch_use_v2_model_runner.py index 756dec9d50ba..00c2c6337afb 100644 --- a/vllm_ascend/patch/platform/patch_use_v2_model_runner.py +++ b/vllm_ascend/patch/platform/patch_use_v2_model_runner.py @@ -1,10 +1,14 @@ -import vllm.envs as envs from vllm.config.vllm import VllmConfig -from vllm_ascend.utils import is_310p, vllm_version_is +from vllm_ascend.mrv2_utils import apply_v2_model_runner_config_patch +from vllm_ascend.utils import vllm_version_is from vllm_ascend.worker.v2.pp_utils import resolve_spec_pp_support -_original_validate_v2_model_runner = VllmConfig._validate_v2_model_runner +# Drive use_v2_model_runner from the Ascend model/feature whitelist instead of +# the env-only override. Also neutralize upstream V2 validation: Ascend owns +# that decision in mrv2_utils (see apply_v2_model_runner_config_patch). +apply_v2_model_runner_config_patch() + _original_get_unsupported_features = VllmConfig._get_v2_model_runner_unsupported_features _ASCEND_V1_SUPPORTED_FEATURES = frozenset( @@ -15,21 +19,6 @@ ) -def _patched_use_v2_model_runner(self) -> bool: - """Return VLLM_USE_V2_MODEL_RUNNER env directly. - - The upstream use_v2_model_runner gate-keeps the v2 runner with - per-model architecture whitelists, Triton availability checks, and - feature-support inspections. On Ascend the v2 runner is controlled - purely by the VLLM_USE_V2_MODEL_RUNNER environment variable; - model-compatibility decisions are deferred to the NPU runner itself. - """ - use_v2 = envs.VLLM_USE_V2_MODEL_RUNNER - if use_v2 is not None: - return use_v2 - return False - - def _patched_get_unsupported_features(self) -> list[str]: unsupported = _original_get_unsupported_features(self) if vllm_version_is("0.28.0") and "prefill context parallelism" in unsupported: @@ -43,20 +32,12 @@ def _patched_get_unsupported_features(self) -> list[str]: return unsupported -VllmConfig.use_v2_model_runner = property(_patched_use_v2_model_runner) VllmConfig._get_v2_model_runner_unsupported_features = _patched_get_unsupported_features - -def _patched_validate_v2_model_runner(self) -> None: - if is_310p(): - return - _original_validate_v2_model_runner(self) - - -VllmConfig._validate_v2_model_runner = _patched_validate_v2_model_runner - -# vLLM main exposes this helper; the supported v0.28.0 lane does not. -if not vllm_version_is("0.28.0"): +# vLLM main exposes this helper; v0.28.0 does not. Prefer hasattr over +# vllm_version_is(): CI installs from a commit SHA can report __version__="dev" +# and would otherwise apply the main-only patch on a release-lane checkout. +if hasattr(VllmConfig, "_get_v1_model_runner_unsupported_features"): _original_get_v1_model_runner_unsupported_features = VllmConfig._get_v1_model_runner_unsupported_features def _patched_get_v1_model_runner_unsupported_features(self) -> list[str]: diff --git a/vllm_ascend/patch/worker/__init__.py b/vllm_ascend/patch/worker/__init__.py index 3809e113dea3..6c778b9bd381 100644 --- a/vllm_ascend/patch/worker/__init__.py +++ b/vllm_ascend/patch/worker/__init__.py @@ -44,10 +44,9 @@ import vllm_ascend.patch.worker.patch_cudagraph # noqa import vllm_ascend.patch.worker.patch_deepseek_v2 # noqa -# vLLM's use_v2_model_runner may enable the v2 runner without the -# VLLM_USE_V2_MODEL_RUNNER env var (e.g. based on model architecture). -# We always patch it so that on Ascend the v2 runner is enabled only -# when the env var is explicitly set. +# Re-apply the Ascend V2 model runner overrides in worker processes +# (whitelist default + remaining V2/V1 feature patches). The env var +# VLLM_USE_V2_MODEL_RUNNER still wins when set. import vllm_ascend.patch.worker.patch_v2.patch_use_v2_model_runner # noqa import vllm_ascend.patch.worker.patch_fused_moe # noqa diff --git a/vllm_ascend/platform.py b/vllm_ascend/platform.py index d1769adae65d..ea4ceafda78a 100644 --- a/vllm_ascend/platform.py +++ b/vllm_ascend/platform.py @@ -38,6 +38,7 @@ QuantizationBackendFamily, get_current_hardware_profile, ) +from vllm_ascend.mrv2_utils import apply_v2_model_runner_config_patch # isort: off from vllm_ascend.utils import ( @@ -441,6 +442,18 @@ def _validate_indexer_pp_config(cls, vllm_config: VllmConfig) -> None: @classmethod def check_and_update_config(cls, vllm_config: VllmConfig) -> None: + # NOTE: This still monkey-patches VllmConfig by replacing the + # use_v2_model_runner property (the "patch way"). It is kept here + # because upstream vLLM does not yet expose a platform hook to + # customize the default V2 model runner decision; the whitelist + # logic itself lives in vllm_ascend.mrv2_utils. + # The upstream V2 validation is also neutralized, since Ascend fully + # owns the V2 enablement decision (the platform / Triton gates in + # mrv2_utils differ from the upstream validation). + # TODO(wxsIcey): Remove this once upstream vLLM allows platforms to + # override the default, and contribute the whitelist upstream. + apply_v2_model_runner_config_patch() + # Lazy import vllm/vllm-ascend to avoid circular import from vllm_ascend.quantization.utils import maybe_auto_detect_quantization from vllm_ascend.logger import configure_ascend_file_logging, configure_ascend_logging diff --git a/vllm_ascend/worker/worker.py b/vllm_ascend/worker/worker.py index dfff01429ce8..35f3fc096e16 100644 --- a/vllm_ascend/worker/worker.py +++ b/vllm_ascend/worker/worker.py @@ -141,9 +141,14 @@ def __init__( "In most scenarios, without custom kernels, vllm-ascend will not function correctly." ) - # register patch for vllm + # Worker processes receive a pickled VllmConfig, so the platform config + # hook (NPUPlatform.check_and_update_config) never runs here. Re-apply the + # Ascend V2 model runner overrides so the worker resolves the same + # runner version as the engine. Idempotent. + from vllm_ascend.mrv2_utils import apply_v2_model_runner_config_patch from vllm_ascend.utils import adapt_patch + apply_v2_model_runner_config_patch() adapt_patch() # Register ops when worker init. From 49c9f3b45a9d076f701ac4e6143d4368bc961cca Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Mon, 14 Sep 2026 11:55:08 +0000 Subject: [PATCH 02/17] test(rlhf): pin sleep/wake e2e to model runner v1 Qwen3ForCausalLM now defaults to MRv2. The RLHF sleep/wake suite still depends on the V1 generate path after CuMem remap, so keep the subprocess server on VLLM_USE_V2_MODEL_RUNNER=0. Signed-off-by: yjyang62 Co-authored-by: yjyang62 --- tests/e2e/pull_request/one_card/rlhf/conftest.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/tests/e2e/pull_request/one_card/rlhf/conftest.py b/tests/e2e/pull_request/one_card/rlhf/conftest.py index 920d9d194ca1..b2b9d71e9ab6 100644 --- a/tests/e2e/pull_request/one_card/rlhf/conftest.py +++ b/tests/e2e/pull_request/one_card/rlhf/conftest.py @@ -94,6 +94,10 @@ def server( **os.environ, "VLLM_SERVER_DEV_MODE": "1", "HF_HUB_OFFLINE": "1", + # Qwen3ForCausalLM now defaults to MRv2. Sleep/wake generate is still + # V1-only; pin the RLHF server to V1 until the MRv2 allocator path is + # ready. Keep this on the subprocess env only. + "VLLM_USE_V2_MODEL_RUNNER": "0", } base = _DUMMY_ARGS if dummy_weights else _BASE_ARGS cmd = [ From 4b97bc5596c39e770800f28e7523fd988beeead1 Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Mon, 14 Sep 2026 16:04:02 +0000 Subject: [PATCH 03/17] fix(platform): isolate MRv2 extras when V2 is whitelist-enabled GPU V2 stores capturing on the forward-context object. Ascend FIA treats _EXTRA_CTX.capturing as ACL graph capture. When Qwen3 defaults to V2 via the whitelist (env unset), extras leaked onto ctx.capturing and graph_task_group_begin ran on a non-capturing stream (error 107029). Route extras through additional_kwargs whenever VllmConfig.use_v2_model_runner is true, not only when VLLM_USE_V2_MODEL_RUNNER is set. Signed-off-by: yjyang62 Co-authored-by: yjyang62 --- tests/ut/test_ascend_forward_context.py | 61 +++++++++++++++++++++++++ vllm_ascend/ascend_forward_context.py | 33 +++++++++++-- 2 files changed, 90 insertions(+), 4 deletions(-) diff --git a/tests/ut/test_ascend_forward_context.py b/tests/ut/test_ascend_forward_context.py index 84fcfb22a3df..1d810f9ab01b 100644 --- a/tests/ut/test_ascend_forward_context.py +++ b/tests/ut/test_ascend_forward_context.py @@ -511,3 +511,64 @@ def fake_set_forward_context(**_kwargs): assert seen["config"] is vllm_config assert seen["inside"] is False + + +def test_extra_ctx_whitelist_v2_hides_gpu_capturing_flag(monkeypatch): + monkeypatch.setattr(afc.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", None) + forward_context = SimpleNamespace( + additional_kwargs={}, + vllm_config=SimpleNamespace(use_v2_model_runner=True), + capturing=True, + ) + monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) + + assert afc._EXTRA_CTX.capturing is None + afc._EXTRA_CTX.capturing = False + assert afc._EXTRA_CTX.capturing is False + assert forward_context.capturing is True + assert forward_context.additional_kwargs["capturing"] is False + + +def test_extra_ctx_v1_stores_capturing_on_context(monkeypatch): + monkeypatch.setattr(afc.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", None) + forward_context = SimpleNamespace( + additional_kwargs={}, + vllm_config=SimpleNamespace(use_v2_model_runner=False), + capturing=False, + ) + monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) + + afc._EXTRA_CTX.capturing = True + assert afc._EXTRA_CTX.capturing is True + assert forward_context.capturing is True + assert "capturing" not in forward_context.additional_kwargs + + +def test_extra_ctx_env_override_wins_over_whitelist(monkeypatch): + monkeypatch.setattr(afc.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", False) + forward_context = SimpleNamespace( + additional_kwargs={}, + vllm_config=SimpleNamespace(use_v2_model_runner=True), + capturing=False, + ) + monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) + + afc._EXTRA_CTX.capturing = True + assert forward_context.capturing is True + assert "capturing" not in forward_context.additional_kwargs + + +def test_extra_ctx_env_true_uses_additional_kwargs(monkeypatch): + monkeypatch.setattr(afc.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", True) + forward_context = SimpleNamespace( + additional_kwargs={}, + vllm_config=SimpleNamespace(use_v2_model_runner=False), + capturing=True, + ) + monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) + + assert afc._EXTRA_CTX.capturing is None + afc._EXTRA_CTX.capturing = False + assert afc._EXTRA_CTX.capturing is False + assert forward_context.capturing is True + assert forward_context.additional_kwargs["capturing"] is False diff --git a/vllm_ascend/ascend_forward_context.py b/vllm_ascend/ascend_forward_context.py index 3acce3f56918..f171157e3777 100644 --- a/vllm_ascend/ascend_forward_context.py +++ b/vllm_ascend/ascend_forward_context.py @@ -420,6 +420,24 @@ def select_moe_comm_method( return moe_comm_type +def _extra_ctx_uses_additional_kwargs(ctx: Any) -> bool: + """Return whether Ascend extras belong in ``ctx.additional_kwargs``. + + GPU V2 stores its own ``capturing`` flag on the forward-context object. + Ascend FIA treats ``_EXTRA_CTX.capturing`` as "this stream is in ACL graph + capture, so call ``graph_task_group_begin``". Those meanings must not mix. + + ``VLLM_USE_V2_MODEL_RUNNER=1`` already isolated extras in + ``additional_kwargs``. The Ascend whitelist can enable V2 with the env + unset, so also follow ``VllmConfig.use_v2_model_runner``. + """ + env = envs_vllm.VLLM_USE_V2_MODEL_RUNNER + if env is not None: + return bool(env) + vllm_config = getattr(ctx, "vllm_config", None) + return bool(getattr(vllm_config, "use_v2_model_runner", False)) + + class _ExtraForwardContextProxy: """Unified forward-context access for v1/v2 model runners.""" @@ -464,17 +482,24 @@ def _ctx(): def __getattr__(self, name: str) -> Any: self.check_extra_attr(name) ctx = self._ctx() - if envs_vllm.VLLM_USE_V2_MODEL_RUNNER: + if _extra_ctx_uses_additional_kwargs(ctx): # Unset known extras default to None so optional flags (e.g. `sinks`) # can be read with truthiness checks before the V2 path populates them. - return ctx.additional_kwargs.get(name) + additional_kwargs = getattr(ctx, "additional_kwargs", None) + if additional_kwargs is None: + return None + return additional_kwargs.get(name) return getattr(ctx, name, None) def __setattr__(self, name: str, value: Any) -> None: self.check_extra_attr(name) ctx = self._ctx() - if envs_vllm.VLLM_USE_V2_MODEL_RUNNER: - ctx.additional_kwargs[name] = value + if _extra_ctx_uses_additional_kwargs(ctx): + additional_kwargs = getattr(ctx, "additional_kwargs", None) + if additional_kwargs is None: + additional_kwargs = {} + ctx.additional_kwargs = additional_kwargs + additional_kwargs[name] = value else: setattr(ctx, name, value) From 32eb36e94a2ffe94fd5d39c39a77929c48aabf3b Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Mon, 14 Sep 2026 16:47:39 +0000 Subject: [PATCH 04/17] fix(platform): ignore MagicMock use_v2_model_runner in extra ctx Whitelist-enabled V2 routed extras through additional_kwargs whenever ctx.vllm_config.use_v2_model_runner was truthy. cpu-ut fixtures pass a bare MagicMock forward context, which auto-creates a truthy flag and hides capturing / max_tokens_across_dp. Require an actual bool True. Signed-off-by: yjyang62 Co-authored-by: yjyang62 --- tests/ut/test_ascend_forward_context.py | 11 +++++++++++ vllm_ascend/ascend_forward_context.py | 5 ++++- 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/tests/ut/test_ascend_forward_context.py b/tests/ut/test_ascend_forward_context.py index 1d810f9ab01b..46632db7e867 100644 --- a/tests/ut/test_ascend_forward_context.py +++ b/tests/ut/test_ascend_forward_context.py @@ -558,6 +558,17 @@ def test_extra_ctx_env_override_wins_over_whitelist(monkeypatch): assert "capturing" not in forward_context.additional_kwargs +def test_extra_ctx_magicmock_forward_context_stays_on_v1_attrs(monkeypatch): + monkeypatch.setattr(afc.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", None) + forward_context = MagicMock(capturing=False) + monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) + + assert afc._extra_ctx_uses_additional_kwargs(forward_context) is False + assert afc._EXTRA_CTX.capturing is False + afc._EXTRA_CTX.capturing = True + assert forward_context.capturing is True + + def test_extra_ctx_env_true_uses_additional_kwargs(monkeypatch): monkeypatch.setattr(afc.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", True) forward_context = SimpleNamespace( diff --git a/vllm_ascend/ascend_forward_context.py b/vllm_ascend/ascend_forward_context.py index f171157e3777..3598720defc3 100644 --- a/vllm_ascend/ascend_forward_context.py +++ b/vllm_ascend/ascend_forward_context.py @@ -435,7 +435,10 @@ def _extra_ctx_uses_additional_kwargs(ctx: Any) -> bool: if env is not None: return bool(env) vllm_config = getattr(ctx, "vllm_config", None) - return bool(getattr(vllm_config, "use_v2_model_runner", False)) + # Require an actual bool. MagicMock forward-context fixtures auto-create a + # truthy use_v2_model_runner and would otherwise hide attrs like capturing + # behind additional_kwargs.get(), which is None. + return getattr(vllm_config, "use_v2_model_runner", False) is True class _ExtraForwardContextProxy: From b0e7077fd84f0ce8892e487bfdd6031a01e34d9d Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Tue, 15 Sep 2026 04:10:58 +0000 Subject: [PATCH 05/17] fix(attention): gate FIA graph_task_group on ACL stream capture GPU V2 sets forward_context.capturing during piecewise warmup. Ascend FIA treated that as ACL capture and called graph_task_group_begin, which failed with 107029 and hung Qwen3 whitelist-MRv2 e2e. Use the live NPU stream capture state (and skip PIECEWISE) instead of _EXTRA_CTX.capturing when wrapping FIA/PA kernels. Signed-off-by: yjyang62 Co-authored-by: yjyang62 --- tests/ut/test_ascend_forward_context.py | 48 +++++++++++++++++++ vllm_ascend/ascend_forward_context.py | 20 ++++++++ .../context_parallel/attention_cp.py | 4 +- .../attention/context_parallel/mla_cp.py | 4 +- vllm_ascend/attention/mla_v1.py | 4 +- 5 files changed, 74 insertions(+), 6 deletions(-) diff --git a/tests/ut/test_ascend_forward_context.py b/tests/ut/test_ascend_forward_context.py index 46632db7e867..3daf22a4f2f2 100644 --- a/tests/ut/test_ascend_forward_context.py +++ b/tests/ut/test_ascend_forward_context.py @@ -583,3 +583,51 @@ def test_extra_ctx_env_true_uses_additional_kwargs(monkeypatch): assert afc._EXTRA_CTX.capturing is False assert forward_context.capturing is True assert forward_context.additional_kwargs["capturing"] is False + + +def test_is_acl_full_graph_capturing_ignores_gpu_flag_when_stream_idle(monkeypatch): + monkeypatch.setattr( + afc.torch, + "npu", + SimpleNamespace(is_current_stream_capturing=lambda: False), + raising=False, + ) + monkeypatch.setattr( + afc, + "get_forward_context", + lambda: SimpleNamespace(cudagraph_runtime_mode=afc.CUDAGraphMode.FULL, capturing=True), + ) + + assert afc.is_acl_full_graph_capturing() is False + + +def test_is_acl_full_graph_capturing_requires_full_mode(monkeypatch): + monkeypatch.setattr( + afc.torch, + "npu", + SimpleNamespace(is_current_stream_capturing=lambda: True), + raising=False, + ) + monkeypatch.setattr( + afc, + "get_forward_context", + lambda: SimpleNamespace(cudagraph_runtime_mode=afc.CUDAGraphMode.FULL), + ) + + assert afc.is_acl_full_graph_capturing() is True + + +def test_is_acl_full_graph_capturing_skips_piecewise(monkeypatch): + monkeypatch.setattr( + afc.torch, + "npu", + SimpleNamespace(is_current_stream_capturing=lambda: True), + raising=False, + ) + monkeypatch.setattr( + afc, + "get_forward_context", + lambda: SimpleNamespace(cudagraph_runtime_mode=afc.CUDAGraphMode.PIECEWISE), + ) + + assert afc.is_acl_full_graph_capturing() is False diff --git a/vllm_ascend/ascend_forward_context.py b/vllm_ascend/ascend_forward_context.py index 3598720defc3..5bfb53553981 100644 --- a/vllm_ascend/ascend_forward_context.py +++ b/vllm_ascend/ascend_forward_context.py @@ -420,6 +420,26 @@ def select_moe_comm_method( return moe_comm_type +def is_acl_full_graph_capturing() -> bool: + """Return whether FIA should wrap kernels in ACL ``graph_task_group``. + + GPU V2 sets ``forward_context.capturing`` during CUDA-graph warmup and + piecewise capture. That flag is not "the current NPU stream is in ACL + capture". Calling ``graph_task_group_begin`` on a non-capturing stream + raises error 107029 and hung Qwen3 whitelist-MRv2 e2e. + + Piecewise graphs must not use FIA task groups. ``ModelWithContext`` already + excludes ``PIECEWISE`` when writing ``_EXTRA_CTX.capturing``; this helper + re-checks the live stream so compiled attention cannot bake in GPU's flag. + """ + npu = getattr(torch, "npu", None) + is_capturing = getattr(npu, "is_current_stream_capturing", None) + if is_capturing is None or not is_capturing(): + return False + ctx = get_forward_context() + return getattr(ctx, "cudagraph_runtime_mode", None) != CUDAGraphMode.PIECEWISE + + def _extra_ctx_uses_additional_kwargs(ctx: Any) -> bool: """Return whether Ascend extras belong in ``ctx.additional_kwargs``. diff --git a/vllm_ascend/attention/context_parallel/attention_cp.py b/vllm_ascend/attention/context_parallel/attention_cp.py index 83011dead585..5650448aceec 100644 --- a/vllm_ascend/attention/context_parallel/attention_cp.py +++ b/vllm_ascend/attention/context_parallel/attention_cp.py @@ -22,7 +22,7 @@ import torch.distributed as dist import torch_npu -from vllm_ascend.ascend_forward_context import _EXTRA_CTX +from vllm_ascend.ascend_forward_context import _EXTRA_CTX, is_acl_full_graph_capturing from vllm_ascend.attention.attention_v1 import ( AscendAttentionBackendImpl, AscendAttentionMetadataBuilder, @@ -343,7 +343,7 @@ def _forward_decode_dcp( else: num_tokens = query.shape[0] * query.shape[1] - if _EXTRA_CTX.capturing: + if is_acl_full_graph_capturing(): stream = torch_npu.npu.current_stream() event = torch.npu.ExternalEvent() diff --git a/vllm_ascend/attention/context_parallel/mla_cp.py b/vllm_ascend/attention/context_parallel/mla_cp.py index 6784679eadc2..92e4e6ff94c0 100644 --- a/vllm_ascend/attention/context_parallel/mla_cp.py +++ b/vllm_ascend/attention/context_parallel/mla_cp.py @@ -23,7 +23,7 @@ ) # isort: on -from vllm_ascend.ascend_forward_context import _EXTRA_CTX +from vllm_ascend.ascend_forward_context import _EXTRA_CTX, is_acl_full_graph_capturing from vllm_ascend.attention.context_parallel.common_cp import ( DCPImplMixin, DCPMetadataBuilderMixin, @@ -714,7 +714,7 @@ def _forward_decode( graph_params = get_draft_graph_params() else: graph_params = get_graph_params() - if _EXTRA_CTX.capturing: + if is_acl_full_graph_capturing(): stream = torch_npu.npu.current_stream() event = torch.npu.ExternalEvent() event.wait(stream) diff --git a/vllm_ascend/attention/mla_v1.py b/vllm_ascend/attention/mla_v1.py index 5871cc49d657..dbee7c454138 100644 --- a/vllm_ascend/attention/mla_v1.py +++ b/vllm_ascend/attention/mla_v1.py @@ -22,7 +22,7 @@ from vllm.v1.kv_cache_interface import AttentionSpec from vllm_ascend.ascend_config import get_ascend_config -from vllm_ascend.ascend_forward_context import _EXTRA_CTX +from vllm_ascend.ascend_forward_context import _EXTRA_CTX, is_acl_full_graph_capturing from vllm_ascend.attention.attention_mask import AttentionMaskBuilder from vllm_ascend.attention.attention_v1 import AscendAttentionState from vllm_ascend.attention.utils import ( @@ -1765,7 +1765,7 @@ def _forward_decode( graph_params = get_draft_graph_params() else: graph_params = get_graph_params() - if _EXTRA_CTX.capturing: + if is_acl_full_graph_capturing(): stream = torch_npu.npu.current_stream() event = torch.npu.ExternalEvent() From 2c0efe5c6f025f88282585c945ed8d467476c792 Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Tue, 15 Sep 2026 04:34:55 +0000 Subject: [PATCH 06/17] fix(attention): treat MagicMock stream capturing as idle cpu-ut stubs torch.npu as MagicMock, so is_current_stream_capturing() is truthy and FIA entered full_graph_fia. Require an actual True, matching use_v2_model_runner. Co-authored-by: yjyang62 Signed-off-by: yjyang62 --- tests/ut/conftest.py | 5 +++++ tests/ut/test_ascend_forward_context.py | 14 ++++++++++++++ vllm_ascend/ascend_forward_context.py | 9 ++++++++- 3 files changed, 27 insertions(+), 1 deletion(-) diff --git a/tests/ut/conftest.py b/tests/ut/conftest.py index df61729d85b6..4d972673aa5f 100644 --- a/tests/ut/conftest.py +++ b/tests/ut/conftest.py @@ -164,6 +164,10 @@ def synchronize(self): torch.npu.graph_task_update_begin = MagicMock() torch.npu.graph_task_update_end = MagicMock() torch.npu.stream = MagicMock() + # cpu-ut is never capturing an ACL graph. Leave this unstubbed and + # is_current_stream_capturing() is a truthy MagicMock, which would send + # FIA into graph_task_group_begin. + torch.npu.is_current_stream_capturing = MagicMock(return_value=False) # Some code paths do `import torch.npu`; attribute assignment alone is not enough. sys.modules["torch.npu"] = torch.npu torch_npu.npu.Stream = _NpuStreamStub # type: ignore[attr-defined] @@ -253,6 +257,7 @@ def synchronize(self): # Re-sync after enable_custom_op / adapt_patch so @patch("torch.npu.*") hits # the same object production code uses via `torch.npu`. torch.npu.current_device = MagicMock(return_value="cpu") + torch.npu.is_current_stream_capturing = MagicMock(return_value=False) sys.modules["torch.npu"] = torch.npu # Clean up any stale mock modules that may have been installed by diff --git a/tests/ut/test_ascend_forward_context.py b/tests/ut/test_ascend_forward_context.py index 3daf22a4f2f2..6e9ba6708af7 100644 --- a/tests/ut/test_ascend_forward_context.py +++ b/tests/ut/test_ascend_forward_context.py @@ -585,6 +585,20 @@ def test_extra_ctx_env_true_uses_additional_kwargs(monkeypatch): assert forward_context.additional_kwargs["capturing"] is False +def test_is_acl_full_graph_capturing_false_for_mock_stream_status(monkeypatch): + monkeypatch.setattr( + afc.torch, + "npu", + SimpleNamespace(is_current_stream_capturing=MagicMock()), + raising=False, + ) + mock_get_ctx = MagicMock() + monkeypatch.setattr(afc, "get_forward_context", mock_get_ctx) + + assert afc.is_acl_full_graph_capturing() is False + mock_get_ctx.assert_not_called() + + def test_is_acl_full_graph_capturing_ignores_gpu_flag_when_stream_idle(monkeypatch): monkeypatch.setattr( afc.torch, diff --git a/vllm_ascend/ascend_forward_context.py b/vllm_ascend/ascend_forward_context.py index 5bfb53553981..808743bf573b 100644 --- a/vllm_ascend/ascend_forward_context.py +++ b/vllm_ascend/ascend_forward_context.py @@ -431,10 +431,17 @@ def is_acl_full_graph_capturing() -> bool: Piecewise graphs must not use FIA task groups. ``ModelWithContext`` already excludes ``PIECEWISE`` when writing ``_EXTRA_CTX.capturing``; this helper re-checks the live stream so compiled attention cannot bake in GPU's flag. + + Require an actual ``True``. cpu-ut stubs ``torch.npu`` as a MagicMock, so + ``is_current_stream_capturing()`` is truthy even when no stream is + capturing. That would send eager FIA UTs into ``full_graph_fia``. """ npu = getattr(torch, "npu", None) is_capturing = getattr(npu, "is_current_stream_capturing", None) - if is_capturing is None or not is_capturing(): + if is_capturing is None: + return False + # MagicMock is truthy; only a real True means the NPU stream is capturing. + if is_capturing() is not True: return False ctx = get_forward_context() return getattr(ctx, "cudagraph_runtime_mode", None) != CUDAGraphMode.PIECEWISE From ab5c4cf388f895b9b088bf4107c10a5c6912c14f Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Tue, 15 Sep 2026 13:20:17 +0000 Subject: [PATCH 07/17] fix(platform): disable default MRv2 for LoRA and dynamic spec decode A2 CI hung on LoRA-only generate and failed dflash acceptance when batch-size-based dynamic K (num_speculative_tokens_per_batch_size) was enabled under the Qwen3 default-V2 whitelist. Drop both from the default feature whitelist; static eagle3/mtp/dflash stay enabled, and VLLM_USE_V2_MODEL_RUNNER still overrides. Signed-off-by: yjyang62 Co-authored-by: yjyang62 --- tests/ut/test_mrv2_utils.py | 102 ++++++++++++++++++++++++++++++++++-- vllm_ascend/mrv2_utils.py | 24 ++++++++- 2 files changed, 121 insertions(+), 5 deletions(-) diff --git a/tests/ut/test_mrv2_utils.py b/tests/ut/test_mrv2_utils.py index 7429185bd1ca..d4fce3c95306 100644 --- a/tests/ut/test_mrv2_utils.py +++ b/tests/ut/test_mrv2_utils.py @@ -44,10 +44,19 @@ def _make_model_config(**kwargs) -> SimpleNamespace: def _make_vllm_config( model_config=None, speculative_config=None, + lora_config=None, ) -> SimpleNamespace: return SimpleNamespace( model_config=model_config, speculative_config=speculative_config, + lora_config=lora_config, + ) + + +def _make_speculative_config(method: str, num_speculative_tokens_per_batch_size=None): + return SimpleNamespace( + method=method, + num_speculative_tokens_per_batch_size=num_speculative_tokens_per_batch_size, ) @@ -94,28 +103,76 @@ def test_without_speculative_config(self): @pytest.mark.parametrize("method", ["eagle3", "mtp", "dflash"]) def test_whitelisted_methods(self, monkeypatch, method): monkeypatch.setattr(mrv2_utils.logger, "info_once", lambda *args: None) - config = _make_vllm_config(speculative_config=SimpleNamespace(method=method)) + config = _make_vllm_config(speculative_config=_make_speculative_config(method)) assert is_supported_v2_model_runner_feature(config) is True @pytest.mark.parametrize("method", ["ngram", "ngram_gpu", "eagle", "unknown_method"]) def test_unsupported_method(self, method): - config = _make_vllm_config(speculative_config=SimpleNamespace(method=method)) + config = _make_vllm_config(speculative_config=_make_speculative_config(method)) assert is_supported_v2_model_runner_feature(config) is False def test_whitelisted_method_logs_info(self, monkeypatch): info_calls = [] monkeypatch.setattr(mrv2_utils.logger, "info_once", lambda *args: info_calls.append(args)) - config = _make_vllm_config(speculative_config=SimpleNamespace(method="eagle3")) + config = _make_vllm_config(speculative_config=_make_speculative_config("eagle3")) assert is_supported_v2_model_runner_feature(config) is True assert len(info_calls) == 1 + def test_lora_is_excluded(self, monkeypatch): + warning_calls = [] + monkeypatch.setattr(mrv2_utils.logger, "warning_once", lambda *args: warning_calls.append(args)) + config = _make_vllm_config(lora_config=object()) + + assert is_supported_v2_model_runner_feature(config) is False + assert len(warning_calls) == 1 + + def test_lora_is_excluded_even_with_whitelisted_spec(self, monkeypatch): + monkeypatch.setattr(mrv2_utils.logger, "warning_once", lambda *args: None) + config = _make_vllm_config( + speculative_config=_make_speculative_config("eagle3"), + lora_config=object(), + ) + + assert is_supported_v2_model_runner_feature(config) is False + + @pytest.mark.parametrize("method", ["eagle3", "mtp", "dflash"]) + def test_dynamic_speculative_decoding_is_excluded(self, monkeypatch, method): + warning_calls = [] + monkeypatch.setattr(mrv2_utils.logger, "warning_once", lambda *args: warning_calls.append(args)) + config = _make_vllm_config( + speculative_config=_make_speculative_config( + method, + num_speculative_tokens_per_batch_size=[[1, 256, 4]], + ) + ) + + assert is_supported_v2_model_runner_feature(config) is False + assert len(warning_calls) == 1 + class TestV2ModelRunnerEnvironmentReady: def test_unsupported_feature(self): - config = _make_vllm_config(speculative_config=SimpleNamespace(method="ngram")) + config = _make_vllm_config(speculative_config=_make_speculative_config("ngram")) + + assert _v2_model_runner_environment_ready(config) is False + + def test_lora_is_not_ready(self, monkeypatch): + monkeypatch.setattr(mrv2_utils.logger, "warning_once", lambda *args: None) + config = _make_vllm_config(lora_config=object()) + + assert _v2_model_runner_environment_ready(config) is False + + def test_dynamic_speculative_decoding_is_not_ready(self, monkeypatch): + monkeypatch.setattr(mrv2_utils.logger, "warning_once", lambda *args: None) + config = _make_vllm_config( + speculative_config=_make_speculative_config( + "dflash", + num_speculative_tokens_per_batch_size=[[1, 256, 4]], + ) + ) assert _v2_model_runner_environment_ready(config) is False @@ -186,6 +243,43 @@ def test_none_model_config_falls_back(self, monkeypatch): assert use_v2_model_runner(config) is False assert len(warning_calls) == 1 + def test_default_disabled_for_lora(self, monkeypatch): + monkeypatch.setattr(mrv2_utils.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", None) + monkeypatch.setattr(mrv2_utils, "is_310p", lambda: False) + monkeypatch.setattr("vllm.triton_utils.HAS_TRITON", True) + monkeypatch.setattr(mrv2_utils.logger, "warning_once", lambda *args: None) + config = _make_vllm_config( + model_config=_make_model_config(architectures=[DEFAULT_V2_ARCH]), + lora_config=object(), + ) + + assert use_v2_model_runner(config) is False + + def test_default_disabled_for_dynamic_speculative_decoding(self, monkeypatch): + monkeypatch.setattr(mrv2_utils.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", None) + monkeypatch.setattr(mrv2_utils, "is_310p", lambda: False) + monkeypatch.setattr("vllm.triton_utils.HAS_TRITON", True) + monkeypatch.setattr(mrv2_utils.logger, "warning_once", lambda *args: None) + monkeypatch.setattr(mrv2_utils.logger, "info_once", lambda *args: None) + config = _make_vllm_config( + model_config=_make_model_config(architectures=[DEFAULT_V2_ARCH]), + speculative_config=_make_speculative_config( + "dflash", + num_speculative_tokens_per_batch_size=[[1, 256, 4]], + ), + ) + + assert use_v2_model_runner(config) is False + + def test_env_override_wins_with_lora(self, monkeypatch): + monkeypatch.setattr(mrv2_utils.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", True) + config = _make_vllm_config( + model_config=_make_model_config(architectures=[DEFAULT_V2_ARCH]), + lora_config=object(), + ) + + assert use_v2_model_runner(config) is True + class TestV2ModelRunnerValidationPatch: def test_validation_is_decoupled_from_upstream(self): diff --git a/vllm_ascend/mrv2_utils.py b/vllm_ascend/mrv2_utils.py index 46a9caa60401..1a813885eeff 100644 --- a/vllm_ascend/mrv2_utils.py +++ b/vllm_ascend/mrv2_utils.py @@ -93,10 +93,32 @@ def is_default_v2_model_runner_model(vllm_config: VllmConfig) -> bool: def is_supported_v2_model_runner_feature(vllm_config: VllmConfig) -> bool: - """Feature whitelist: only whitelisted features may be enabled with a whitelisted model.""" + """Feature whitelist: only whitelisted features may be enabled with a whitelisted model. + + LoRA and batch-size-based dynamic speculative decoding + (``num_speculative_tokens_per_batch_size``) are excluded from the + default-V2 feature whitelist. Static ``eagle3`` / ``mtp`` / ``dflash`` + remain supported. ``VLLM_USE_V2_MODEL_RUNNER`` still overrides this + default decision. + """ + if getattr(vllm_config, "lora_config", None) is not None: + logger.warning_once( + "Model Runner V2 default is disabled because LoRA is enabled; using the V1 model runner instead." + ) + return False + speculative_config = vllm_config.speculative_config if speculative_config is None: return True + + if getattr(speculative_config, "num_speculative_tokens_per_batch_size", None): + logger.warning_once( + "Model Runner V2 default is disabled because dynamic speculative " + "decoding (num_speculative_tokens_per_batch_size) is enabled; " + "using the V1 model runner instead." + ) + return False + if speculative_config.method in ("eagle3", "mtp", "dflash"): logger.info_once( "Model Runner V2 is enabled by default for speculative method '%s'.", From 4605ad5be762e7bcf14b3ac31f23ebae4bfab99e Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Tue, 15 Sep 2026 16:42:05 +0000 Subject: [PATCH 08/17] fix(worker): map CUDA stream capturing to NPU in MRv2 GPU V2 PrefetchOffloader calls torch.cuda.is_current_stream_capturing during load_model. V1 already remaps that CUDA dummy to torch.npu; V2 torch_cuda_wrapper did not, so Qwen3 default-V2 prefetch e2e crashed on NPU. Signed-off-by: yjyang62 --- tests/ut/worker/v2/test_v2_utils.py | 4 ++++ vllm_ascend/worker/v2/utils.py | 2 ++ 2 files changed, 6 insertions(+) diff --git a/tests/ut/worker/v2/test_v2_utils.py b/tests/ut/worker/v2/test_v2_utils.py index 6e5066bd5db2..549b0cca9d97 100644 --- a/tests/ut/worker/v2/test_v2_utils.py +++ b/tests/ut/worker/v2/test_v2_utils.py @@ -33,7 +33,11 @@ def test_v2_utils_context_managers_switch_and_restore(): with v2_utils.torch_cuda_wrapper(): assert fake_cuda.Event is fake_npu.Event assert fake_cuda.graph is v2_utils.torch_npu_graph_wrapper + assert fake_cuda.is_current_stream_capturing is fake_npu.is_current_stream_capturing assert v2_utils.breakable_cudagraph.weak_ref_tensor is v2_utils.weak_ref_tensor + # Mapping is process-wide and must survive runner init so GPU V2 + # load_model / PrefetchOffloader can query capture state on NPU. + assert fake_cuda.is_current_stream_capturing is fake_npu.is_current_stream_capturing with v2_utils.communicator_switch(): assert cuda_comm_mod.CudaCommunicator is npu_cls diff --git a/vllm_ascend/worker/v2/utils.py b/vllm_ascend/worker/v2/utils.py index 7fd3cf7e5996..a8b0483b080c 100644 --- a/vllm_ascend/worker/v2/utils.py +++ b/vllm_ascend/worker/v2/utils.py @@ -24,6 +24,8 @@ def torch_cuda_wrapper(): torch.cuda.set_stream = torch.npu.set_stream torch.cuda.current_device = torch.npu.current_device torch.cuda.mem_get_info = torch.npu.mem_get_info + # GPU V2 prefetch/offload calls this CUDA API during load_model. + torch.cuda.is_current_stream_capturing = torch.npu.is_current_stream_capturing breakable_cudagraph.weak_ref_tensor = weak_ref_tensor breakable_cudagraph.weak_ref_tensors = weak_ref_tensors logger.info_once("Wrapping torch.cuda with torch.npu.") From 0ddd7f1409ee80e4eb84250a2c1e487f77bde770 Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Wed, 16 Sep 2026 01:30:34 +0000 Subject: [PATCH 09/17] test(e2e): drop NZ-graph prefetch xfail under default MRv2 Qwen3ForCausalLM now defaults to Model Runner V2. GPU V2 prefetch plus NZ graph capture matches the eager baseline, so the strict xfail on test_prefetch_offload_accuracy[NZ-graph] XPASS-fails a2-1 CI. Keep the V1 AscendPrefetchOffloader fail-fast for the unsupported combo. Signed-off-by: yjyang62 --- .../one_card/test_cpu_weight_offload.py | 20 +++---------------- 1 file changed, 3 insertions(+), 17 deletions(-) diff --git a/tests/e2e/pull_request/one_card/test_cpu_weight_offload.py b/tests/e2e/pull_request/one_card/test_cpu_weight_offload.py index 0afb4b4a59bc..73006ceafb7b 100644 --- a/tests/e2e/pull_request/one_card/test_cpu_weight_offload.py +++ b/tests/e2e/pull_request/one_card/test_cpu_weight_offload.py @@ -94,23 +94,9 @@ def _compare_offload_logprobs( pytest.param(True, 0, id="ND-eager"), pytest.param(False, 0, id="ND-graph"), pytest.param(True, 2, id="NZ-eager"), - # TODO(wangfiox): nz+graph not supported yet - pytest.param( - False, - 2, - id="NZ-graph", - marks=pytest.mark.xfail( - strict=True, - reason=( - "NZ static buffers make the prefetch H2D copy a " - "cross-format (ND->NZ) conversion that is aclop-only on " - "CANN 9.0.0 and rejected during ACL graph capture; " - "AscendPrefetchOffloader fails fast with a clear error " - "for this combo. Remove this marker and the offloader " - "guard once the no-transdata prefetch path lands." - ), - ), - ), + # Qwen3 defaults to MRv2; GPU V2 prefetch + NZ graph now matches the + # eager baseline. V1 still fail-fasts in AscendPrefetchOffloader. + pytest.param(False, 2, id="NZ-graph"), ], ) @wait_until_npu_memory_free() From 669d37811168e30c872c59e105fa71ffc5f9ef16 Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Wed, 16 Sep 2026 03:20:18 +0000 Subject: [PATCH 10/17] fix(platform): isolate MRv2 extras via current VllmConfig Route _EXTRA_CTX through additional_kwargs when use_v2_model_runner(get_current_vllm_config()) is true. Whitelist-default V2 leaves VLLM_USE_V2_MODEL_RUNNER unset and GPU ForwardContext has no vllm_config, so env/ctx checks leaked GPU capturing onto FIA. Restore FIA/PA graph_task_group gating to _EXTRA_CTX.capturing. Signed-off-by: yjyang62 --- tests/ut/test_ascend_forward_context.py | 88 ++++--------------- vllm_ascend/ascend_forward_context.py | 60 ++++--------- .../context_parallel/attention_cp.py | 4 +- .../attention/context_parallel/mla_cp.py | 4 +- vllm_ascend/attention/mla_v1.py | 4 +- 5 files changed, 39 insertions(+), 121 deletions(-) diff --git a/tests/ut/test_ascend_forward_context.py b/tests/ut/test_ascend_forward_context.py index 6e9ba6708af7..cea82544bc35 100644 --- a/tests/ut/test_ascend_forward_context.py +++ b/tests/ut/test_ascend_forward_context.py @@ -103,7 +103,8 @@ def test_deepseek_v4_forward_passes_input_ids_to_layers(monkeypatch): from vllm_ascend.models.deepseek_v4 import model as deepseek_v4 - monkeypatch.setattr(afc.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", True) + monkeypatch.setattr(afc, "get_current_vllm_config", lambda: SimpleNamespace()) + monkeypatch.setattr(afc, "use_v2_model_runner", lambda _cfg: True) monkeypatch.setattr( deepseek_v4, "get_pp_group", @@ -514,10 +515,12 @@ def fake_set_forward_context(**_kwargs): def test_extra_ctx_whitelist_v2_hides_gpu_capturing_flag(monkeypatch): - monkeypatch.setattr(afc.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", None) + # GPU V2 ForwardContext has no vllm_config. Isolation must follow + # use_v2_model_runner(get_current_vllm_config()), not ctx.vllm_config. + monkeypatch.setattr(afc, "get_current_vllm_config", lambda: SimpleNamespace()) + monkeypatch.setattr(afc, "use_v2_model_runner", lambda _cfg: True) forward_context = SimpleNamespace( additional_kwargs={}, - vllm_config=SimpleNamespace(use_v2_model_runner=True), capturing=True, ) monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) @@ -530,10 +533,10 @@ def test_extra_ctx_whitelist_v2_hides_gpu_capturing_flag(monkeypatch): def test_extra_ctx_v1_stores_capturing_on_context(monkeypatch): - monkeypatch.setattr(afc.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", None) + monkeypatch.setattr(afc, "get_current_vllm_config", lambda: SimpleNamespace()) + monkeypatch.setattr(afc, "use_v2_model_runner", lambda _cfg: False) forward_context = SimpleNamespace( additional_kwargs={}, - vllm_config=SimpleNamespace(use_v2_model_runner=False), capturing=False, ) monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) @@ -545,10 +548,10 @@ def test_extra_ctx_v1_stores_capturing_on_context(monkeypatch): def test_extra_ctx_env_override_wins_over_whitelist(monkeypatch): - monkeypatch.setattr(afc.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", False) + monkeypatch.setattr(afc, "get_current_vllm_config", lambda: SimpleNamespace()) + monkeypatch.setattr(afc, "use_v2_model_runner", lambda _cfg: False) forward_context = SimpleNamespace( additional_kwargs={}, - vllm_config=SimpleNamespace(use_v2_model_runner=True), capturing=False, ) monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) @@ -559,7 +562,10 @@ def test_extra_ctx_env_override_wins_over_whitelist(monkeypatch): def test_extra_ctx_magicmock_forward_context_stays_on_v1_attrs(monkeypatch): - monkeypatch.setattr(afc.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", None) + def _unset_config(): + raise RuntimeError("Current vllm config is not set.") + + monkeypatch.setattr(afc, "get_current_vllm_config", _unset_config) forward_context = MagicMock(capturing=False) monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) @@ -570,10 +576,10 @@ def test_extra_ctx_magicmock_forward_context_stays_on_v1_attrs(monkeypatch): def test_extra_ctx_env_true_uses_additional_kwargs(monkeypatch): - monkeypatch.setattr(afc.envs_vllm, "VLLM_USE_V2_MODEL_RUNNER", True) + monkeypatch.setattr(afc, "get_current_vllm_config", lambda: SimpleNamespace()) + monkeypatch.setattr(afc, "use_v2_model_runner", lambda _cfg: True) forward_context = SimpleNamespace( additional_kwargs={}, - vllm_config=SimpleNamespace(use_v2_model_runner=False), capturing=True, ) monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) @@ -583,65 +589,3 @@ def test_extra_ctx_env_true_uses_additional_kwargs(monkeypatch): assert afc._EXTRA_CTX.capturing is False assert forward_context.capturing is True assert forward_context.additional_kwargs["capturing"] is False - - -def test_is_acl_full_graph_capturing_false_for_mock_stream_status(monkeypatch): - monkeypatch.setattr( - afc.torch, - "npu", - SimpleNamespace(is_current_stream_capturing=MagicMock()), - raising=False, - ) - mock_get_ctx = MagicMock() - monkeypatch.setattr(afc, "get_forward_context", mock_get_ctx) - - assert afc.is_acl_full_graph_capturing() is False - mock_get_ctx.assert_not_called() - - -def test_is_acl_full_graph_capturing_ignores_gpu_flag_when_stream_idle(monkeypatch): - monkeypatch.setattr( - afc.torch, - "npu", - SimpleNamespace(is_current_stream_capturing=lambda: False), - raising=False, - ) - monkeypatch.setattr( - afc, - "get_forward_context", - lambda: SimpleNamespace(cudagraph_runtime_mode=afc.CUDAGraphMode.FULL, capturing=True), - ) - - assert afc.is_acl_full_graph_capturing() is False - - -def test_is_acl_full_graph_capturing_requires_full_mode(monkeypatch): - monkeypatch.setattr( - afc.torch, - "npu", - SimpleNamespace(is_current_stream_capturing=lambda: True), - raising=False, - ) - monkeypatch.setattr( - afc, - "get_forward_context", - lambda: SimpleNamespace(cudagraph_runtime_mode=afc.CUDAGraphMode.FULL), - ) - - assert afc.is_acl_full_graph_capturing() is True - - -def test_is_acl_full_graph_capturing_skips_piecewise(monkeypatch): - monkeypatch.setattr( - afc.torch, - "npu", - SimpleNamespace(is_current_stream_capturing=lambda: True), - raising=False, - ) - monkeypatch.setattr( - afc, - "get_forward_context", - lambda: SimpleNamespace(cudagraph_runtime_mode=afc.CUDAGraphMode.PIECEWISE), - ) - - assert afc.is_acl_full_graph_capturing() is False diff --git a/vllm_ascend/ascend_forward_context.py b/vllm_ascend/ascend_forward_context.py index 808743bf573b..ad53cc90bbc6 100644 --- a/vllm_ascend/ascend_forward_context.py +++ b/vllm_ascend/ascend_forward_context.py @@ -5,8 +5,7 @@ from typing import Any import torch -import vllm.envs as envs_vllm -from vllm.config import CUDAGraphMode, VllmConfig, set_current_vllm_config +from vllm.config import CUDAGraphMode, VllmConfig, get_current_vllm_config, set_current_vllm_config from vllm.distributed import get_dp_group, get_ep_group, get_tensor_model_parallel_world_size from vllm.forward_context import BatchDescriptor, get_forward_context, set_forward_context from vllm.logger import logger @@ -17,6 +16,7 @@ MoECommPolicy, get_current_hardware_profile, ) +from vllm_ascend.mrv2_utils import use_v2_model_runner from vllm_ascend.quantization.quant_type import A5_SUPPORT_MEGA_MOE_QUANT_TYPES, QuantType from vllm_ascend.utils import ( has_layer_idx, @@ -420,52 +420,26 @@ def select_moe_comm_method( return moe_comm_type -def is_acl_full_graph_capturing() -> bool: - """Return whether FIA should wrap kernels in ACL ``graph_task_group``. - - GPU V2 sets ``forward_context.capturing`` during CUDA-graph warmup and - piecewise capture. That flag is not "the current NPU stream is in ACL - capture". Calling ``graph_task_group_begin`` on a non-capturing stream - raises error 107029 and hung Qwen3 whitelist-MRv2 e2e. - - Piecewise graphs must not use FIA task groups. ``ModelWithContext`` already - excludes ``PIECEWISE`` when writing ``_EXTRA_CTX.capturing``; this helper - re-checks the live stream so compiled attention cannot bake in GPU's flag. - - Require an actual ``True``. cpu-ut stubs ``torch.npu`` as a MagicMock, so - ``is_current_stream_capturing()`` is truthy even when no stream is - capturing. That would send eager FIA UTs into ``full_graph_fia``. - """ - npu = getattr(torch, "npu", None) - is_capturing = getattr(npu, "is_current_stream_capturing", None) - if is_capturing is None: - return False - # MagicMock is truthy; only a real True means the NPU stream is capturing. - if is_capturing() is not True: - return False - ctx = get_forward_context() - return getattr(ctx, "cudagraph_runtime_mode", None) != CUDAGraphMode.PIECEWISE - - -def _extra_ctx_uses_additional_kwargs(ctx: Any) -> bool: +def _extra_ctx_uses_additional_kwargs(_ctx: Any) -> bool: """Return whether Ascend extras belong in ``ctx.additional_kwargs``. GPU V2 stores its own ``capturing`` flag on the forward-context object. - Ascend FIA treats ``_EXTRA_CTX.capturing`` as "this stream is in ACL graph - capture, so call ``graph_task_group_begin``". Those meanings must not mix. + Ascend FIA treats ``_EXTRA_CTX.capturing`` as ACL graph capture. Isolate + extras whenever Ascend enables V2, including the architecture whitelist + with ``VLLM_USE_V2_MODEL_RUNNER`` unset. - ``VLLM_USE_V2_MODEL_RUNNER=1`` already isolated extras in - ``additional_kwargs``. The Ascend whitelist can enable V2 with the env - unset, so also follow ``VllmConfig.use_v2_model_runner``. + Use ``get_current_vllm_config()`` rather than ``ctx.vllm_config``: GPU V2 + ``ForwardContext`` has no ``vllm_config`` field, so whitelist-default V2 + would otherwise leak GPU's ``capturing`` onto ``_EXTRA_CTX``. """ - env = envs_vllm.VLLM_USE_V2_MODEL_RUNNER - if env is not None: - return bool(env) - vllm_config = getattr(ctx, "vllm_config", None) - # Require an actual bool. MagicMock forward-context fixtures auto-create a - # truthy use_v2_model_runner and would otherwise hide attrs like capturing - # behind additional_kwargs.get(), which is None. - return getattr(vllm_config, "use_v2_model_runner", False) is True + try: + vllm_config = get_current_vllm_config() + except Exception: + return False + if vllm_config is None: + return False + # Require an actual bool. MagicMock configs can be truthy. + return use_v2_model_runner(vllm_config) is True class _ExtraForwardContextProxy: diff --git a/vllm_ascend/attention/context_parallel/attention_cp.py b/vllm_ascend/attention/context_parallel/attention_cp.py index 5650448aceec..83011dead585 100644 --- a/vllm_ascend/attention/context_parallel/attention_cp.py +++ b/vllm_ascend/attention/context_parallel/attention_cp.py @@ -22,7 +22,7 @@ import torch.distributed as dist import torch_npu -from vllm_ascend.ascend_forward_context import _EXTRA_CTX, is_acl_full_graph_capturing +from vllm_ascend.ascend_forward_context import _EXTRA_CTX from vllm_ascend.attention.attention_v1 import ( AscendAttentionBackendImpl, AscendAttentionMetadataBuilder, @@ -343,7 +343,7 @@ def _forward_decode_dcp( else: num_tokens = query.shape[0] * query.shape[1] - if is_acl_full_graph_capturing(): + if _EXTRA_CTX.capturing: stream = torch_npu.npu.current_stream() event = torch.npu.ExternalEvent() diff --git a/vllm_ascend/attention/context_parallel/mla_cp.py b/vllm_ascend/attention/context_parallel/mla_cp.py index 92e4e6ff94c0..6784679eadc2 100644 --- a/vllm_ascend/attention/context_parallel/mla_cp.py +++ b/vllm_ascend/attention/context_parallel/mla_cp.py @@ -23,7 +23,7 @@ ) # isort: on -from vllm_ascend.ascend_forward_context import _EXTRA_CTX, is_acl_full_graph_capturing +from vllm_ascend.ascend_forward_context import _EXTRA_CTX from vllm_ascend.attention.context_parallel.common_cp import ( DCPImplMixin, DCPMetadataBuilderMixin, @@ -714,7 +714,7 @@ def _forward_decode( graph_params = get_draft_graph_params() else: graph_params = get_graph_params() - if is_acl_full_graph_capturing(): + if _EXTRA_CTX.capturing: stream = torch_npu.npu.current_stream() event = torch.npu.ExternalEvent() event.wait(stream) diff --git a/vllm_ascend/attention/mla_v1.py b/vllm_ascend/attention/mla_v1.py index dbee7c454138..5871cc49d657 100644 --- a/vllm_ascend/attention/mla_v1.py +++ b/vllm_ascend/attention/mla_v1.py @@ -22,7 +22,7 @@ from vllm.v1.kv_cache_interface import AttentionSpec from vllm_ascend.ascend_config import get_ascend_config -from vllm_ascend.ascend_forward_context import _EXTRA_CTX, is_acl_full_graph_capturing +from vllm_ascend.ascend_forward_context import _EXTRA_CTX from vllm_ascend.attention.attention_mask import AttentionMaskBuilder from vllm_ascend.attention.attention_v1 import AscendAttentionState from vllm_ascend.attention.utils import ( @@ -1765,7 +1765,7 @@ def _forward_decode( graph_params = get_draft_graph_params() else: graph_params = get_graph_params() - if is_acl_full_graph_capturing(): + if _EXTRA_CTX.capturing: stream = torch_npu.npu.current_stream() event = torch.npu.ExternalEvent() From 25b7c9ec64c385c9dc9a9a030508d580f6c9f771 Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Wed, 16 Sep 2026 03:54:17 +0000 Subject: [PATCH 11/17] refactor(platform): inline extra-ctx V2 isolation check Replace `_extra_ctx_uses_additional_kwargs` with `if use_v2_model_runner(get_current_vllm_config()) is True`. Guard xlite `index_full_mask` for mypy after merging main. Signed-off-by: yjyang62 --- tests/ut/test_ascend_forward_context.py | 1 - vllm_ascend/ascend_forward_context.py | 56 +++++++++---------------- vllm_ascend/xlite/xlite.py | 2 +- 3 files changed, 20 insertions(+), 39 deletions(-) diff --git a/tests/ut/test_ascend_forward_context.py b/tests/ut/test_ascend_forward_context.py index cea82544bc35..7116b6162f32 100644 --- a/tests/ut/test_ascend_forward_context.py +++ b/tests/ut/test_ascend_forward_context.py @@ -569,7 +569,6 @@ def _unset_config(): forward_context = MagicMock(capturing=False) monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) - assert afc._extra_ctx_uses_additional_kwargs(forward_context) is False assert afc._EXTRA_CTX.capturing is False afc._EXTRA_CTX.capturing = True assert forward_context.capturing is True diff --git a/vllm_ascend/ascend_forward_context.py b/vllm_ascend/ascend_forward_context.py index ad53cc90bbc6..9151369b671c 100644 --- a/vllm_ascend/ascend_forward_context.py +++ b/vllm_ascend/ascend_forward_context.py @@ -420,28 +420,6 @@ def select_moe_comm_method( return moe_comm_type -def _extra_ctx_uses_additional_kwargs(_ctx: Any) -> bool: - """Return whether Ascend extras belong in ``ctx.additional_kwargs``. - - GPU V2 stores its own ``capturing`` flag on the forward-context object. - Ascend FIA treats ``_EXTRA_CTX.capturing`` as ACL graph capture. Isolate - extras whenever Ascend enables V2, including the architecture whitelist - with ``VLLM_USE_V2_MODEL_RUNNER`` unset. - - Use ``get_current_vllm_config()`` rather than ``ctx.vllm_config``: GPU V2 - ``ForwardContext`` has no ``vllm_config`` field, so whitelist-default V2 - would otherwise leak GPU's ``capturing`` onto ``_EXTRA_CTX``. - """ - try: - vllm_config = get_current_vllm_config() - except Exception: - return False - if vllm_config is None: - return False - # Require an actual bool. MagicMock configs can be truthy. - return use_v2_model_runner(vllm_config) is True - - class _ExtraForwardContextProxy: """Unified forward-context access for v1/v2 model runners.""" @@ -486,26 +464,30 @@ def _ctx(): def __getattr__(self, name: str) -> Any: self.check_extra_attr(name) ctx = self._ctx() - if _extra_ctx_uses_additional_kwargs(ctx): - # Unset known extras default to None so optional flags (e.g. `sinks`) - # can be read with truthiness checks before the V2 path populates them. - additional_kwargs = getattr(ctx, "additional_kwargs", None) - if additional_kwargs is None: - return None - return additional_kwargs.get(name) + try: + if use_v2_model_runner(get_current_vllm_config()) is True: + additional_kwargs = getattr(ctx, "additional_kwargs", None) + if additional_kwargs is None: + return None + return additional_kwargs.get(name) + except Exception: + pass return getattr(ctx, name, None) def __setattr__(self, name: str, value: Any) -> None: self.check_extra_attr(name) ctx = self._ctx() - if _extra_ctx_uses_additional_kwargs(ctx): - additional_kwargs = getattr(ctx, "additional_kwargs", None) - if additional_kwargs is None: - additional_kwargs = {} - ctx.additional_kwargs = additional_kwargs - additional_kwargs[name] = value - else: - setattr(ctx, name, value) + try: + if use_v2_model_runner(get_current_vllm_config()) is True: + additional_kwargs = getattr(ctx, "additional_kwargs", None) + if additional_kwargs is None: + additional_kwargs = {} + ctx.additional_kwargs = additional_kwargs + additional_kwargs[name] = value + return + except Exception: + pass + setattr(ctx, name, value) # usage: from vllm_ascend.ascend_forward_context import _EXTRA_CTX diff --git a/vllm_ascend/xlite/xlite.py b/vllm_ascend/xlite/xlite.py index f71e4d22fdb1..e39d58fef150 100644 --- a/vllm_ascend/xlite/xlite.py +++ b/vllm_ascend/xlite/xlite.py @@ -661,7 +661,7 @@ def extract_kv_cache(self, kv_caches: list[tuple[torch.Tensor, ...]], /) -> list mla_caches: list[tuple[torch.Tensor, ...]] = [] idx = 0 - index_mask = self.xlite_config.index_full_mask or [True] * self.xlite_config.n_layers + index_mask = getattr(self.xlite_config, "index_full_mask", None) or [True] * self.xlite_config.n_layers dummy_indexer_cache = (_DUMMY_TENSOR,) for mask in index_mask: indexer_caches.append(kv_caches[idx] if mask else dummy_indexer_cache) From 676ebe730f299ba814e663965f133ba531e58a5a Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Wed, 16 Sep 2026 03:58:46 +0000 Subject: [PATCH 12/17] refactor(platform): drop extra-ctx try around V2 check Keep getattr/setattr the same as before, only replacing the helper with `if use_v2_model_runner(get_current_vllm_config()) is True`. Signed-off-by: yjyang62 --- tests/ut/test_ascend_forward_context.py | 6 ++--- vllm_ascend/ascend_forward_context.py | 34 +++++++++++-------------- 2 files changed, 17 insertions(+), 23 deletions(-) diff --git a/tests/ut/test_ascend_forward_context.py b/tests/ut/test_ascend_forward_context.py index 7116b6162f32..181adcb7d03a 100644 --- a/tests/ut/test_ascend_forward_context.py +++ b/tests/ut/test_ascend_forward_context.py @@ -562,10 +562,8 @@ def test_extra_ctx_env_override_wins_over_whitelist(monkeypatch): def test_extra_ctx_magicmock_forward_context_stays_on_v1_attrs(monkeypatch): - def _unset_config(): - raise RuntimeError("Current vllm config is not set.") - - monkeypatch.setattr(afc, "get_current_vllm_config", _unset_config) + monkeypatch.setattr(afc, "get_current_vllm_config", lambda: MagicMock()) + monkeypatch.setattr(afc, "use_v2_model_runner", lambda _cfg: MagicMock()) forward_context = MagicMock(capturing=False) monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) diff --git a/vllm_ascend/ascend_forward_context.py b/vllm_ascend/ascend_forward_context.py index 9151369b671c..2292546aba08 100644 --- a/vllm_ascend/ascend_forward_context.py +++ b/vllm_ascend/ascend_forward_context.py @@ -464,30 +464,26 @@ def _ctx(): def __getattr__(self, name: str) -> Any: self.check_extra_attr(name) ctx = self._ctx() - try: - if use_v2_model_runner(get_current_vllm_config()) is True: - additional_kwargs = getattr(ctx, "additional_kwargs", None) - if additional_kwargs is None: - return None - return additional_kwargs.get(name) - except Exception: - pass + if use_v2_model_runner(get_current_vllm_config()) is True: + # Unset known extras default to None so optional flags (e.g. `sinks`) + # can be read with truthiness checks before the V2 path populates them. + additional_kwargs = getattr(ctx, "additional_kwargs", None) + if additional_kwargs is None: + return None + return additional_kwargs.get(name) return getattr(ctx, name, None) def __setattr__(self, name: str, value: Any) -> None: self.check_extra_attr(name) ctx = self._ctx() - try: - if use_v2_model_runner(get_current_vllm_config()) is True: - additional_kwargs = getattr(ctx, "additional_kwargs", None) - if additional_kwargs is None: - additional_kwargs = {} - ctx.additional_kwargs = additional_kwargs - additional_kwargs[name] = value - return - except Exception: - pass - setattr(ctx, name, value) + if use_v2_model_runner(get_current_vllm_config()) is True: + additional_kwargs = getattr(ctx, "additional_kwargs", None) + if additional_kwargs is None: + additional_kwargs = {} + ctx.additional_kwargs = additional_kwargs + additional_kwargs[name] = value + else: + setattr(ctx, name, value) # usage: from vllm_ascend.ascend_forward_context import _EXTRA_CTX From 3538c1092da6054eeddb8bb8e027995d8f38ca71 Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Wed, 16 Sep 2026 04:00:55 +0000 Subject: [PATCH 13/17] refactor(platform): restore original extra-ctx kwargs access Keep getattr/setattr the same as main. Only replace the V2 condition with `use_v2_model_runner(get_current_vllm_config()) is True`. Signed-off-by: yjyang62 --- vllm_ascend/ascend_forward_context.py | 11 ++--------- 1 file changed, 2 insertions(+), 9 deletions(-) diff --git a/vllm_ascend/ascend_forward_context.py b/vllm_ascend/ascend_forward_context.py index 2292546aba08..14de1850e637 100644 --- a/vllm_ascend/ascend_forward_context.py +++ b/vllm_ascend/ascend_forward_context.py @@ -467,21 +467,14 @@ def __getattr__(self, name: str) -> Any: if use_v2_model_runner(get_current_vllm_config()) is True: # Unset known extras default to None so optional flags (e.g. `sinks`) # can be read with truthiness checks before the V2 path populates them. - additional_kwargs = getattr(ctx, "additional_kwargs", None) - if additional_kwargs is None: - return None - return additional_kwargs.get(name) + return ctx.additional_kwargs.get(name) return getattr(ctx, name, None) def __setattr__(self, name: str, value: Any) -> None: self.check_extra_attr(name) ctx = self._ctx() if use_v2_model_runner(get_current_vllm_config()) is True: - additional_kwargs = getattr(ctx, "additional_kwargs", None) - if additional_kwargs is None: - additional_kwargs = {} - ctx.additional_kwargs = additional_kwargs - additional_kwargs[name] = value + ctx.additional_kwargs[name] = value else: setattr(ctx, name, value) From 05275967c4de6b82f311b4f6fbfa0ab0c0ab5240 Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Wed, 16 Sep 2026 04:58:35 +0000 Subject: [PATCH 14/17] fix(platform): treat unset VllmConfig as V1 extra-ctx get_current_vllm_config() raises in cpu-ut attention fixtures. Fall back to V1 extra-ctx attrs so FIA can read capturing. Signed-off-by: yjyang62 Co-authored-by: yjyang62 --- tests/ut/test_ascend_forward_context.py | 13 +++++++++++++ vllm_ascend/ascend_forward_context.py | 22 ++++++++++++++++++++-- 2 files changed, 33 insertions(+), 2 deletions(-) diff --git a/tests/ut/test_ascend_forward_context.py b/tests/ut/test_ascend_forward_context.py index 181adcb7d03a..2637b2ec62d3 100644 --- a/tests/ut/test_ascend_forward_context.py +++ b/tests/ut/test_ascend_forward_context.py @@ -572,6 +572,19 @@ def test_extra_ctx_magicmock_forward_context_stays_on_v1_attrs(monkeypatch): assert forward_context.capturing is True +def test_extra_ctx_unset_vllm_config_stays_on_v1_attrs(monkeypatch): + def _unset_config(): + raise AssertionError("Current vLLM config is not set.") + + monkeypatch.setattr(afc, "get_current_vllm_config", _unset_config) + forward_context = MagicMock(capturing=False) + monkeypatch.setattr(afc, "get_forward_context", lambda: forward_context) + + assert afc._EXTRA_CTX.capturing is False + afc._EXTRA_CTX.capturing = True + assert forward_context.capturing is True + + def test_extra_ctx_env_true_uses_additional_kwargs(monkeypatch): monkeypatch.setattr(afc, "get_current_vllm_config", lambda: SimpleNamespace()) monkeypatch.setattr(afc, "use_v2_model_runner", lambda _cfg: True) diff --git a/vllm_ascend/ascend_forward_context.py b/vllm_ascend/ascend_forward_context.py index 14de1850e637..e970d17542ab 100644 --- a/vllm_ascend/ascend_forward_context.py +++ b/vllm_ascend/ascend_forward_context.py @@ -420,6 +420,24 @@ def select_moe_comm_method( return moe_comm_type +def _use_v2_extra_kwargs() -> bool: + """Return whether Ascend extras belong in ``ctx.additional_kwargs``. + + GPU V2 stores its own ``capturing`` flag on the forward-context object. + Isolate extras when Ascend enables V2, including whitelist-default V2 + with ``VLLM_USE_V2_MODEL_RUNNER`` unset. + + ``get_current_vllm_config()`` raises when no config is set (cpu-ut + attention fixtures). Treat that as V1 so FIA can read ``capturing``. + Require an actual bool: MagicMock configs can be truthy. + """ + try: + vllm_config = get_current_vllm_config() + except Exception: + return False + return use_v2_model_runner(vllm_config) is True + + class _ExtraForwardContextProxy: """Unified forward-context access for v1/v2 model runners.""" @@ -464,7 +482,7 @@ def _ctx(): def __getattr__(self, name: str) -> Any: self.check_extra_attr(name) ctx = self._ctx() - if use_v2_model_runner(get_current_vllm_config()) is True: + if _use_v2_extra_kwargs(): # Unset known extras default to None so optional flags (e.g. `sinks`) # can be read with truthiness checks before the V2 path populates them. return ctx.additional_kwargs.get(name) @@ -473,7 +491,7 @@ def __getattr__(self, name: str) -> Any: def __setattr__(self, name: str, value: Any) -> None: self.check_extra_attr(name) ctx = self._ctx() - if use_v2_model_runner(get_current_vllm_config()) is True: + if _use_v2_extra_kwargs(): ctx.additional_kwargs[name] = value else: setattr(ctx, name, value) From bd955ec8ef8ad81640b11f29615deb5f4c33b441 Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Wed, 16 Sep 2026 06:15:32 +0000 Subject: [PATCH 15/17] fix(platform): keep extra-ctx V2 isolation out of Dynamo Compiled FIA/MoE read _EXTRA_CTX, which called use_v2_model_runner() and hit logger.warning_once / info_once. Dynamo cannot trace those logs, so LoRA and non-whitelist models failed compile, and V2 MoE baked additional_kwargs.get("moe_comm_method") as None. Disable Dynamo on the extras helper, proxy getattr/setattr, and use_v2_model_runner so isolation still uses get_current_vllm_config at runtime. Signed-off-by: yjyang62 Co-authored-by: yjyang62 --- tests/ut/test_ascend_forward_context.py | 8 ++++++++ tests/ut/test_mrv2_utils.py | 3 +++ vllm_ascend/ascend_forward_context.py | 6 ++++++ vllm_ascend/mrv2_utils.py | 2 ++ 4 files changed, 19 insertions(+) diff --git a/tests/ut/test_ascend_forward_context.py b/tests/ut/test_ascend_forward_context.py index 2637b2ec62d3..e02e0205b5ac 100644 --- a/tests/ut/test_ascend_forward_context.py +++ b/tests/ut/test_ascend_forward_context.py @@ -514,6 +514,14 @@ def fake_set_forward_context(**_kwargs): assert seen["inside"] is False +def test_extra_ctx_v2_isolation_is_dynamo_disabled(): + # Compiled attention/MoE read _EXTRA_CTX. Dynamo cannot trace + # use_v2_model_runner's logger.warning_once / info_once. + assert getattr(afc._use_v2_extra_kwargs, "_dynamo_disable", False) + assert getattr(afc._ExtraForwardContextProxy.__getattr__, "_dynamo_disable", False) + assert getattr(afc._ExtraForwardContextProxy.__setattr__, "_dynamo_disable", False) + + def test_extra_ctx_whitelist_v2_hides_gpu_capturing_flag(monkeypatch): # GPU V2 ForwardContext has no vllm_config. Isolation must follow # use_v2_model_runner(get_current_vllm_config()), not ctx.vllm_config. diff --git a/tests/ut/test_mrv2_utils.py b/tests/ut/test_mrv2_utils.py index d4fce3c95306..c05e18f6bf98 100644 --- a/tests/ut/test_mrv2_utils.py +++ b/tests/ut/test_mrv2_utils.py @@ -282,6 +282,9 @@ def test_env_override_wins_with_lora(self, monkeypatch): class TestV2ModelRunnerValidationPatch: + def test_use_v2_model_runner_is_dynamo_disabled(self): + assert getattr(mrv2_utils.use_v2_model_runner, "_dynamo_disable", False) + def test_validation_is_decoupled_from_upstream(self): # The Ascend V2 runner decision is fully owned by use_v2_model_runner, # so the replacement validation must never raise (e.g. the upstream diff --git a/vllm_ascend/ascend_forward_context.py b/vllm_ascend/ascend_forward_context.py index e970d17542ab..2ed3f3677b39 100644 --- a/vllm_ascend/ascend_forward_context.py +++ b/vllm_ascend/ascend_forward_context.py @@ -420,6 +420,7 @@ def select_moe_comm_method( return moe_comm_type +@torch._dynamo.disable def _use_v2_extra_kwargs() -> bool: """Return whether Ascend extras belong in ``ctx.additional_kwargs``. @@ -430,6 +431,9 @@ def _use_v2_extra_kwargs() -> bool: ``get_current_vllm_config()`` raises when no config is set (cpu-ut attention fixtures). Treat that as V1 so FIA can read ``capturing``. Require an actual bool: MagicMock configs can be truthy. + + Disabled under Dynamo: ``use_v2_model_runner`` logs with + ``warning_once`` / ``info_once``, which Dynamo cannot trace. """ try: vllm_config = get_current_vllm_config() @@ -479,6 +483,7 @@ def check_extra_attr(self, name: str): def _ctx(): return get_forward_context() + @torch._dynamo.disable def __getattr__(self, name: str) -> Any: self.check_extra_attr(name) ctx = self._ctx() @@ -488,6 +493,7 @@ def __getattr__(self, name: str) -> Any: return ctx.additional_kwargs.get(name) return getattr(ctx, name, None) + @torch._dynamo.disable def __setattr__(self, name: str, value: Any) -> None: self.check_extra_attr(name) ctx = self._ctx() diff --git a/vllm_ascend/mrv2_utils.py b/vllm_ascend/mrv2_utils.py index 1a813885eeff..bf98bd9db196 100644 --- a/vllm_ascend/mrv2_utils.py +++ b/vllm_ascend/mrv2_utils.py @@ -19,6 +19,7 @@ from typing import TYPE_CHECKING +import torch import vllm.envs as envs_vllm from vllm.logger import logger @@ -146,6 +147,7 @@ def _v2_model_runner_environment_ready(vllm_config: VllmConfig) -> bool: return True +@torch._dynamo.disable def use_v2_model_runner(vllm_config: VllmConfig) -> bool: """Return whether the V2 model runner should be used on Ascend. From a2bc395f7247b921f7ef76efc69e1f5b51cf0bc2 Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Wed, 16 Sep 2026 06:30:21 +0000 Subject: [PATCH 16/17] fix(platform): keep extra-ctx __getattr__ visible to mypy Decorating __getattr__/__setattr__ with torch._dynamo.disable made mypy treat _EXTRA_CTX as having no dynamic attributes. Keep the dunders undecorated and disable the helpers they call instead. Signed-off-by: yjyang62 Co-authored-by: yjyang62 --- tests/ut/test_ascend_forward_context.py | 4 +-- vllm_ascend/ascend_forward_context.py | 40 +++++++++++++++---------- 2 files changed, 26 insertions(+), 18 deletions(-) diff --git a/tests/ut/test_ascend_forward_context.py b/tests/ut/test_ascend_forward_context.py index e02e0205b5ac..b1e29e195107 100644 --- a/tests/ut/test_ascend_forward_context.py +++ b/tests/ut/test_ascend_forward_context.py @@ -518,8 +518,8 @@ def test_extra_ctx_v2_isolation_is_dynamo_disabled(): # Compiled attention/MoE read _EXTRA_CTX. Dynamo cannot trace # use_v2_model_runner's logger.warning_once / info_once. assert getattr(afc._use_v2_extra_kwargs, "_dynamo_disable", False) - assert getattr(afc._ExtraForwardContextProxy.__getattr__, "_dynamo_disable", False) - assert getattr(afc._ExtraForwardContextProxy.__setattr__, "_dynamo_disable", False) + assert getattr(afc._extra_ctx_getattr, "_dynamo_disable", False) + assert getattr(afc._extra_ctx_setattr, "_dynamo_disable", False) def test_extra_ctx_whitelist_v2_hides_gpu_capturing_flag(monkeypatch): diff --git a/vllm_ascend/ascend_forward_context.py b/vllm_ascend/ascend_forward_context.py index 2ed3f3677b39..bc2641d2039f 100644 --- a/vllm_ascend/ascend_forward_context.py +++ b/vllm_ascend/ascend_forward_context.py @@ -483,24 +483,32 @@ def check_extra_attr(self, name: str): def _ctx(): return get_forward_context() - @torch._dynamo.disable def __getattr__(self, name: str) -> Any: - self.check_extra_attr(name) - ctx = self._ctx() - if _use_v2_extra_kwargs(): - # Unset known extras default to None so optional flags (e.g. `sinks`) - # can be read with truthiness checks before the V2 path populates them. - return ctx.additional_kwargs.get(name) - return getattr(ctx, name, None) - - @torch._dynamo.disable + return _extra_ctx_getattr(self, name) + def __setattr__(self, name: str, value: Any) -> None: - self.check_extra_attr(name) - ctx = self._ctx() - if _use_v2_extra_kwargs(): - ctx.additional_kwargs[name] = value - else: - setattr(ctx, name, value) + _extra_ctx_setattr(self, name, value) + + +@torch._dynamo.disable +def _extra_ctx_getattr(proxy: _ExtraForwardContextProxy, name: str) -> Any: + proxy.check_extra_attr(name) + ctx = proxy._ctx() + if _use_v2_extra_kwargs(): + # Unset known extras default to None so optional flags (e.g. `sinks`) + # can be read with truthiness checks before the V2 path populates them. + return ctx.additional_kwargs.get(name) + return getattr(ctx, name, None) + + +@torch._dynamo.disable +def _extra_ctx_setattr(proxy: _ExtraForwardContextProxy, name: str, value: Any) -> None: + proxy.check_extra_attr(name) + ctx = proxy._ctx() + if _use_v2_extra_kwargs(): + ctx.additional_kwargs[name] = value + else: + setattr(ctx, name, value) # usage: from vllm_ascend.ascend_forward_context import _EXTRA_CTX From 4c03f47963c5d629fe96339e6debe69c4ef09de5 Mon Sep 17 00:00:00 2001 From: yjyang62 Date: Wed, 16 Sep 2026 06:47:12 +0000 Subject: [PATCH 17/17] test(platform): assert torch 2.10 Dynamo disable tag cpu-ut installs torch 2.10, which marks @torch._dynamo.disable with _torchdynamo_disable instead of _dynamo_disable. Signed-off-by: yjyang62 Co-authored-by: yjyang62 --- tests/ut/test_ascend_forward_context.py | 11 ++++++++--- tests/ut/test_mrv2_utils.py | 7 ++++++- 2 files changed, 14 insertions(+), 4 deletions(-) diff --git a/tests/ut/test_ascend_forward_context.py b/tests/ut/test_ascend_forward_context.py index b1e29e195107..245e93415e59 100644 --- a/tests/ut/test_ascend_forward_context.py +++ b/tests/ut/test_ascend_forward_context.py @@ -514,12 +514,17 @@ def fake_set_forward_context(**_kwargs): assert seen["inside"] is False +def _is_dynamo_disabled(fn) -> bool: + # torch 2.10 tags `_torchdynamo_disable`; older torch used `_dynamo_disable`. + return bool(getattr(fn, "_torchdynamo_disable", False) or getattr(fn, "_dynamo_disable", False)) + + def test_extra_ctx_v2_isolation_is_dynamo_disabled(): # Compiled attention/MoE read _EXTRA_CTX. Dynamo cannot trace # use_v2_model_runner's logger.warning_once / info_once. - assert getattr(afc._use_v2_extra_kwargs, "_dynamo_disable", False) - assert getattr(afc._extra_ctx_getattr, "_dynamo_disable", False) - assert getattr(afc._extra_ctx_setattr, "_dynamo_disable", False) + assert _is_dynamo_disabled(afc._use_v2_extra_kwargs) + assert _is_dynamo_disabled(afc._extra_ctx_getattr) + assert _is_dynamo_disabled(afc._extra_ctx_setattr) def test_extra_ctx_whitelist_v2_hides_gpu_capturing_flag(monkeypatch): diff --git a/tests/ut/test_mrv2_utils.py b/tests/ut/test_mrv2_utils.py index c05e18f6bf98..cbb547f2497b 100644 --- a/tests/ut/test_mrv2_utils.py +++ b/tests/ut/test_mrv2_utils.py @@ -281,9 +281,14 @@ def test_env_override_wins_with_lora(self, monkeypatch): assert use_v2_model_runner(config) is True +def _is_dynamo_disabled(fn) -> bool: + # torch 2.10 tags `_torchdynamo_disable`; older torch used `_dynamo_disable`. + return bool(getattr(fn, "_torchdynamo_disable", False) or getattr(fn, "_dynamo_disable", False)) + + class TestV2ModelRunnerValidationPatch: def test_use_v2_model_runner_is_dynamo_disabled(self): - assert getattr(mrv2_utils.use_v2_model_runner, "_dynamo_disable", False) + assert _is_dynamo_disabled(mrv2_utils.use_v2_model_runner) def test_validation_is_decoupled_from_upstream(self): # The Ascend V2 runner decision is fully owned by use_v2_model_runner,