diff --git a/gateway/slash_commands.py b/gateway/slash_commands.py index c02c8c1d8a55c..9761fa5af338c 100644 --- a/gateway/slash_commands.py +++ b/gateway/slash_commands.py @@ -3871,13 +3871,20 @@ def _apply_fast_selection(value: str, persist: bool = False) -> Optional[str]: if persist: # ``--global`` lane (upstream scoping): the persisted # agent.service_tier default applies to the configured - # gateway model, so gate on that model's fast support - # rather than the current session route. + # gateway route, so gate on that route rather than the + # current session route. from gateway.run import _resolve_gateway_model - from hermes_cli.models import model_supports_fast_mode - if not model_supports_fast_mode( - _resolve_gateway_model(user_config) - ): + from hermes_cli.models import ( + resolve_fast_mode_capability_for_configured_route, + ) + _, global_provider, global_api_mode = ( + self._configured_route_identity(user_config) + ) + if not resolve_fast_mode_capability_for_configured_route( + model=_resolve_gateway_model(user_config), + provider=global_provider, + api_mode=global_api_mode, + ).supported: return t("gateway.fast.not_supported") else: if persisted_preference and preference_unavailable: @@ -3930,17 +3937,24 @@ def _apply_fast_selection(value: str, persist: bool = False) -> Optional[str]: if not args or args == "status": # Interactive picker on platforms that support it (parity with the # /model + /reasoning pickers). Gated on the configured gateway - # model's fast support; falls through to the text status card when + # route's fast support; falls through to the text status card when # the platform has no picker or the send fails. Session-scoped by # default; `/fast --global` persists agent.service_tier. from gateway.run import _resolve_gateway_model - from hermes_cli.models import model_supports_fast_mode + from hermes_cli.models import ( + resolve_fast_mode_capability_for_configured_route, + ) _fast_supported = False try: - _fast_supported = model_supports_fast_mode( - _resolve_gateway_model(user_config) + _, _cfg_provider, _cfg_api_mode = self._configured_route_identity( + user_config ) + _fast_supported = resolve_fast_mode_capability_for_configured_route( + model=_resolve_gateway_model(user_config), + provider=_cfg_provider, + api_mode=_cfg_api_mode, + ).supported except Exception: _fast_supported = False diff --git a/hermes_cli/models.py b/hermes_cli/models.py index 88fdeaa4be095..bfc9214293726 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -2498,6 +2498,62 @@ def resolve_fast_mode_capability( ) +# A provider that is absent or still ``auto`` has not been pinned to a +# transport yet, so a route built from it cannot be evaluated as a real one. +_UNRESOLVED_PROVIDERS = frozenset({"", "auto"}) + +# The documented native route that carries Fast for each capability family. +# The catalogs are disjoint by model, so at most one entry can match. +_NATIVE_FAST_ROUTES: tuple[tuple[str, str, str], ...] = ( + ("anthropic_fast", "anthropic", "anthropic_messages"), + ("openai_priority", "openai-api", "codex_responses"), +) + + +def _native_fast_route(model_id: Optional[str]) -> Optional[tuple[str, str]]: + """Return the ``(provider, api_mode)`` whose contract documents Fast for a model.""" + base = _fast_model_base(model_id) + if not base: + return None + for family, provider, api_mode in _NATIVE_FAST_ROUTES: + if base in FAST_MODE_CAPABILITY_CATALOG[family]["models"]: + return provider, api_mode + return None + + +def resolve_fast_mode_capability_for_configured_route( + *, + model: Optional[str], + provider: Optional[str], + api_mode: Optional[str], +) -> FastModeCapability: + """Resolve Fast support for a configured route whose provider may be unpinned. + + Config surfaces name a model before its transport is necessarily pinned: + ``model.provider`` may be absent, or still ``auto``. That is not the same + as naming a *wrong* provider, so it must not fail closed the way + :func:`resolve_fast_mode_capability` deliberately does for unknown + providers and proxies — doing so would hide ``/fast`` from a default + config whose model genuinely supports it. + + A concrete provider therefore resolves through the full route contract + unchanged. An unresolved one resolves against the model's own documented + native route, which is the answer a model-only gate reports. + """ + if str(provider or "").strip().lower() not in _UNRESOLVED_PROVIDERS: + return resolve_fast_mode_capability( + model=model, provider=provider, api_mode=api_mode + ) + native = _native_fast_route(model) + if native is None: + return resolve_fast_mode_capability( + model=model, provider=provider, api_mode=api_mode + ) + return resolve_fast_mode_capability( + model=model, provider=native[0], api_mode=native[1] + ) + + def model_supports_fast_mode(model_id: Optional[str]) -> bool: """Return whether Hermes should expose the /fast toggle for this model.""" return _is_anthropic_fast_model(model_id) or _is_openai_fast_model(model_id) diff --git a/tests/cli/test_fast_route_capability.py b/tests/cli/test_fast_route_capability.py index 14d14f08a8142..100cdf9cc0075 100644 --- a/tests/cli/test_fast_route_capability.py +++ b/tests/cli/test_fast_route_capability.py @@ -198,19 +198,6 @@ def test_anthropic_resolver_and_adapter_share_one_immutable_exact_catalog(): assert _supports_fast_mode(model) is expected -@pytest.mark.xfail( - strict=False, - reason=( - "INHERITED (not a merge regression): this AST call-site enforcement fails " - "identically on fork/main 37fa0a353 — verified on a clean baseline worktree " - "during the 2026-07-23 parity sync. It demands that every request-enforcement " - "call site adopt the route-aware resolve_fast_mode_capability() instead of the " - "model-only model_supports_fast_mode() wrapper; that wiring is a real feature " - "task on fork/main, not something this sync introduced or should smuggle in. " - "Follow-up card: parity-doctrine board (route-aware fast-mode call-site " - "adoption). Remove this marker with that work." - ), -) def test_request_enforcement_call_sites_do_not_use_model_only_wrapper(): root = Path(__file__).resolve().parents[2] compatibility_definition = root / "hermes_cli" / "models.py" diff --git a/tests/gateway/test_choice_picker.py b/tests/gateway/test_choice_picker.py index eb68b9eeeeffe..94a2fa8012fde 100644 --- a/tests/gateway/test_choice_picker.py +++ b/tests/gateway/test_choice_picker.py @@ -170,11 +170,19 @@ async def test_picker_show_choice_toggles_display(self, tmp_path, monkeypatch): class TestFastChoicePicker: def _patch_fast_support(self, monkeypatch, tmp_path): + """Configure a route that genuinely carries Fast. + + The gate resolves the configured route's real capability, so this + supplies a real Fast-capable route rather than stubbing the gate — + the picker is then exercised against the same contract production + enforces. + """ monkeypatch.setattr(gateway_run, "_hermes_home", tmp_path) - monkeypatch.setattr(gateway_run, "_load_gateway_config", lambda: {}) - monkeypatch.setattr(gateway_run, "_resolve_gateway_model", lambda cfg: "gpt-5.6") - import hermes_cli.models as models_mod - monkeypatch.setattr(models_mod, "model_supports_fast_mode", lambda m: True) + monkeypatch.setattr( + gateway_run, + "_load_gateway_config", + lambda: {"model": {"default": "gpt-5.5", "provider": "openai-api"}}, + ) @pytest.mark.asyncio async def test_bare_fast_sends_picker_when_adapter_supports_it(self, tmp_path, monkeypatch): diff --git a/tests/run_agent/test_moa_loop_mode.py b/tests/run_agent/test_moa_loop_mode.py index 903db30c21f53..1cb2762137f38 100644 --- a/tests/run_agent/test_moa_loop_mode.py +++ b/tests/run_agent/test_moa_loop_mode.py @@ -1051,41 +1051,48 @@ def fake_call_llm(**kwargs): assert calls[-1]["task"] == "moa_aggregator" -@pytest.mark.xfail( - strict=False, - reason=( - "LOAD-SENSITIVE TIMING (flake, not a regression): asserts advisor calls " - "overlap in wall-clock; green in isolation (verified 3x locally and on " - "several CI rounds during the 2026-07-23 parity sync) but intermittently " - "red under CI's 12-worker parallelism where scheduling can serialize the " - "coroutines. strict=False so a green run still passes. Proper fix is to " - "assert concurrency via an ordering/barrier signal instead of elapsed " - "time — follow-up card on the parity-doctrine board." - ), -) def test_references_run_in_parallel(monkeypatch): """References fan out concurrently (delegate-batch semantics), not serially. - Each reference sleeps; wall-time must approximate the slowest single call, - not the sum. Order is preserved and a failing reference is isolated. + Order is preserved and a failing reference is isolated. """ - import time + import threading from agent import moa_loop # Force _extract_text down its fallback path (no transport normalize). monkeypatch.setattr(moa_loop, "get_transport", lambda *_a, **_k: None) - barrier_hits = [] + # Concurrency is proven by construction, not by measuring elapsed time. + # + # The old form timed the fan-out and asserted the total came in under the + # serial floor of the sleeps. That makes the OS scheduler part of the + # assertion: under a loaded CI box (12 parallel workers) the runtime can + # serialize the worker threads and the inequality flips, so the test went + # red without anything being wrong. It also silently assumed ~zero fixed + # setup before dispatch, which stopped holding once `_run_reference` began + # probing each reference model's context window. + # + # A barrier asserts the invariant directly: both sleeping references must + # be inside `call_llm` AT THE SAME TIME before either is allowed to return. + # If the fan-out ever runs serially the first reference blocks forever + # waiting for a partner that will not arrive, `wait()` raises BrokenBarrier + # on timeout, and the test fails with an explicit message. No wall-clock + # constant, no load sensitivity, and still a hard failure on serialization. + SLEEPERS = 2 + rendezvous = threading.Barrier(SLEEPERS) + overlapped = threading.Event() + entered: list[str] = [] def slow_call_llm(**kwargs): model = kwargs["model"] + entered.append(model) if model == "boom": - barrier_hits.append(("enter", model, time.monotonic())) raise RuntimeError("kaboom") - barrier_hits.append(("enter", model, time.monotonic())) - time.sleep(0.5) - barrier_hits.append(("exit", model, time.monotonic())) + # Generous relative to real scheduling latency, but finite so a serial + # regression fails fast instead of hanging the suite. + rendezvous.wait(timeout=30) + overlapped.set() return _response(f"resp-{kwargs['provider']}") monkeypatch.setattr(moa_loop, "call_llm", slow_call_llm) @@ -1097,45 +1104,22 @@ def slow_call_llm(**kwargs): {"provider": "p3", "model": "ok"}, ] - start = time.monotonic() - out = moa_loop._run_references_parallel( - refs, [{"role": "user", "content": "hi"}], temperature=0.6, max_tokens=64 - ) - elapsed = time.monotonic() - start + try: + out = moa_loop._run_references_parallel( + refs, [{"role": "user", "content": "hi"}], temperature=0.6, max_tokens=64 + ) + except threading.BrokenBarrierError: # pragma: no cover - serial regression + pytest.fail( + "references did not run in parallel: a reference reached the " + "rendezvous alone, so the fan-out never had " + f"{SLEEPERS} references inside call_llm at once (entered: {entered})" + ) - # Assert the INVARIANT (concurrency), not a wall-clock constant. - # - # The old form asserted `elapsed < 0.95`, i.e. "total wall time is under the - # 1.0s serial floor of two 0.5s sleeps". That silently assumed ~zero fixed - # setup before dispatch. It no longer holds: `_run_reference` now calls the - # upstream-new `_trim_messages_for_reference` (#60345 — trims the advisory - # request to the REFERENCE model's own window), whose - # `get_model_context_length` probe costs ~0.4s per distinct (provider, - # model) for models with no cached metadata (as here — "ok"/"boom" are - # fabricated). That overhead is itself incurred CONCURRENTLY in the worker - # threads, so the fan-out is still fully parallel — but `elapsed` becomes - # ~0.4 + 0.5 ≈ 0.95s+ and trips a threshold that was never measuring setup. - # - # The real contract is that the sleeping calls OVERLAP. Assert that - # directly from the timestamps the test already collects: the second - # reference must enter `call_llm` before the first one exits. This is exact, - # immune to fixed setup cost and machine load, and still fails hard if the - # references ever run serially. - sleepers = [h for h in barrier_hits if h[1] == "ok"] - enters = sorted(t for kind, _model, t in sleepers if kind == "enter") - exits = sorted(t for kind, _model, t in sleepers if kind == "exit") - assert len(enters) == 2 and len(exits) == 2, f"expected 2 sleeping refs, got {barrier_hits}" - assert enters[1] < exits[0], ( - "references did not run in parallel: the second reference entered " - f"call_llm at +{enters[1] - start:.2f}s, after the first exited at " - f"+{exits[0] - start:.2f}s (total {elapsed:.2f}s)" - ) - # Sanity bound: overlapping 0.5s sleeps must never approach the 1.0s serial - # floor of sleep time itself. Measured from the first sleeper's entry so - # per-reference setup (the context-window probe above) is excluded. - assert exits[-1] - enters[0] < 0.95, ( - f"sleep phase took {exits[-1] - enters[0]:.2f}s — at/over the serial floor" + # The barrier only clears when both sleepers are inside call_llm together. + assert overlapped.is_set(), ( + f"references never overlapped inside call_llm (entered: {entered})" ) + assert entered.count("ok") == SLEEPERS, f"expected 2 sleeping refs, got {entered}" # Output order matches input order (stable Reference N labelling). assert [label for label, _, _ in out] == ["p1:ok", "moa:preset", "p2:boom", "p3:ok"] assert "recursively reference MoA" in out[1][1]