Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
34 changes: 24 additions & 10 deletions gateway/slash_commands.py
Original file line number Diff line number Diff line change
Expand Up @@ -3871,13 +3871,20 @@ def _apply_fast_selection(value: str, persist: bool = False) -> Optional[str]:
if persist:
# ``--global`` lane (upstream scoping): the persisted
# agent.service_tier default applies to the configured
# gateway model, so gate on that model's fast support
# rather than the current session route.
# gateway route, so gate on that route rather than the
# current session route.
from gateway.run import _resolve_gateway_model
from hermes_cli.models import model_supports_fast_mode
if not model_supports_fast_mode(
_resolve_gateway_model(user_config)
):
from hermes_cli.models import (
resolve_fast_mode_capability_for_configured_route,
)
_, global_provider, global_api_mode = (
self._configured_route_identity(user_config)
)
if not resolve_fast_mode_capability_for_configured_route(
model=_resolve_gateway_model(user_config),
provider=global_provider,
api_mode=global_api_mode,
).supported:
return t("gateway.fast.not_supported")
else:
if persisted_preference and preference_unavailable:
Expand Down Expand Up @@ -3930,17 +3937,24 @@ def _apply_fast_selection(value: str, persist: bool = False) -> Optional[str]:
if not args or args == "status":
# Interactive picker on platforms that support it (parity with the
# /model + /reasoning pickers). Gated on the configured gateway
# model's fast support; falls through to the text status card when
# route's fast support; falls through to the text status card when
# the platform has no picker or the send fails. Session-scoped by
# default; `/fast --global` persists agent.service_tier.
from gateway.run import _resolve_gateway_model
from hermes_cli.models import model_supports_fast_mode
from hermes_cli.models import (
resolve_fast_mode_capability_for_configured_route,
)

_fast_supported = False
try:
_fast_supported = model_supports_fast_mode(
_resolve_gateway_model(user_config)
_, _cfg_provider, _cfg_api_mode = self._configured_route_identity(
user_config
)
_fast_supported = resolve_fast_mode_capability_for_configured_route(
model=_resolve_gateway_model(user_config),
provider=_cfg_provider,
api_mode=_cfg_api_mode,
).supported
except Exception:
_fast_supported = False

Expand Down
56 changes: 56 additions & 0 deletions hermes_cli/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -2498,6 +2498,62 @@ def resolve_fast_mode_capability(
)


# A provider that is absent or still ``auto`` has not been pinned to a
# transport yet, so a route built from it cannot be evaluated as a real one.
_UNRESOLVED_PROVIDERS = frozenset({"", "auto"})

# The documented native route that carries Fast for each capability family.
# The catalogs are disjoint by model, so at most one entry can match.
_NATIVE_FAST_ROUTES: tuple[tuple[str, str, str], ...] = (
("anthropic_fast", "anthropic", "anthropic_messages"),
("openai_priority", "openai-api", "codex_responses"),
)


def _native_fast_route(model_id: Optional[str]) -> Optional[tuple[str, str]]:
"""Return the ``(provider, api_mode)`` whose contract documents Fast for a model."""
base = _fast_model_base(model_id)
if not base:
return None
for family, provider, api_mode in _NATIVE_FAST_ROUTES:
if base in FAST_MODE_CAPABILITY_CATALOG[family]["models"]:
return provider, api_mode
return None


def resolve_fast_mode_capability_for_configured_route(
*,
model: Optional[str],
provider: Optional[str],
api_mode: Optional[str],
) -> FastModeCapability:
"""Resolve Fast support for a configured route whose provider may be unpinned.

Config surfaces name a model before its transport is necessarily pinned:
``model.provider`` may be absent, or still ``auto``. That is not the same
as naming a *wrong* provider, so it must not fail closed the way
:func:`resolve_fast_mode_capability` deliberately does for unknown
providers and proxies — doing so would hide ``/fast`` from a default
config whose model genuinely supports it.

A concrete provider therefore resolves through the full route contract
unchanged. An unresolved one resolves against the model's own documented
native route, which is the answer a model-only gate reports.
"""
if str(provider or "").strip().lower() not in _UNRESOLVED_PROVIDERS:
return resolve_fast_mode_capability(
model=model, provider=provider, api_mode=api_mode
)
native = _native_fast_route(model)
if native is None:
return resolve_fast_mode_capability(
model=model, provider=provider, api_mode=api_mode
)
return resolve_fast_mode_capability(
model=model, provider=native[0], api_mode=native[1]
)


def model_supports_fast_mode(model_id: Optional[str]) -> bool:
"""Return whether Hermes should expose the /fast toggle for this model."""
return _is_anthropic_fast_model(model_id) or _is_openai_fast_model(model_id)
Expand Down
13 changes: 0 additions & 13 deletions tests/cli/test_fast_route_capability.py
Original file line number Diff line number Diff line change
Expand Up @@ -198,19 +198,6 @@ def test_anthropic_resolver_and_adapter_share_one_immutable_exact_catalog():
assert _supports_fast_mode(model) is expected


@pytest.mark.xfail(
strict=False,
reason=(
"INHERITED (not a merge regression): this AST call-site enforcement fails "
"identically on fork/main 37fa0a353 — verified on a clean baseline worktree "
"during the 2026-07-23 parity sync. It demands that every request-enforcement "
"call site adopt the route-aware resolve_fast_mode_capability() instead of the "
"model-only model_supports_fast_mode() wrapper; that wiring is a real feature "
"task on fork/main, not something this sync introduced or should smuggle in. "
"Follow-up card: parity-doctrine board (route-aware fast-mode call-site "
"adoption). Remove this marker with that work."
),
)
def test_request_enforcement_call_sites_do_not_use_model_only_wrapper():
root = Path(__file__).resolve().parents[2]
compatibility_definition = root / "hermes_cli" / "models.py"
Expand Down
16 changes: 12 additions & 4 deletions tests/gateway/test_choice_picker.py
Original file line number Diff line number Diff line change
Expand Up @@ -170,11 +170,19 @@ async def test_picker_show_choice_toggles_display(self, tmp_path, monkeypatch):

class TestFastChoicePicker:
def _patch_fast_support(self, monkeypatch, tmp_path):
"""Configure a route that genuinely carries Fast.

The gate resolves the configured route's real capability, so this
supplies a real Fast-capable route rather than stubbing the gate —
the picker is then exercised against the same contract production
enforces.
"""
monkeypatch.setattr(gateway_run, "_hermes_home", tmp_path)
monkeypatch.setattr(gateway_run, "_load_gateway_config", lambda: {})
monkeypatch.setattr(gateway_run, "_resolve_gateway_model", lambda cfg: "gpt-5.6")
import hermes_cli.models as models_mod
monkeypatch.setattr(models_mod, "model_supports_fast_mode", lambda m: True)
monkeypatch.setattr(
gateway_run,
"_load_gateway_config",
lambda: {"model": {"default": "gpt-5.5", "provider": "openai-api"}},
)

@pytest.mark.asyncio
async def test_bare_fast_sends_picker_when_adapter_supports_it(self, tmp_path, monkeypatch):
Expand Down
98 changes: 41 additions & 57 deletions tests/run_agent/test_moa_loop_mode.py
Original file line number Diff line number Diff line change
Expand Up @@ -1051,41 +1051,48 @@ def fake_call_llm(**kwargs):
assert calls[-1]["task"] == "moa_aggregator"


@pytest.mark.xfail(
strict=False,
reason=(
"LOAD-SENSITIVE TIMING (flake, not a regression): asserts advisor calls "
"overlap in wall-clock; green in isolation (verified 3x locally and on "
"several CI rounds during the 2026-07-23 parity sync) but intermittently "
"red under CI's 12-worker parallelism where scheduling can serialize the "
"coroutines. strict=False so a green run still passes. Proper fix is to "
"assert concurrency via an ordering/barrier signal instead of elapsed "
"time — follow-up card on the parity-doctrine board."
),
)
def test_references_run_in_parallel(monkeypatch):
"""References fan out concurrently (delegate-batch semantics), not serially.

Each reference sleeps; wall-time must approximate the slowest single call,
not the sum. Order is preserved and a failing reference is isolated.
Order is preserved and a failing reference is isolated.
"""
import time
import threading

from agent import moa_loop

# Force _extract_text down its fallback path (no transport normalize).
monkeypatch.setattr(moa_loop, "get_transport", lambda *_a, **_k: None)

barrier_hits = []
# Concurrency is proven by construction, not by measuring elapsed time.
#
# The old form timed the fan-out and asserted the total came in under the
# serial floor of the sleeps. That makes the OS scheduler part of the
# assertion: under a loaded CI box (12 parallel workers) the runtime can
# serialize the worker threads and the inequality flips, so the test went
# red without anything being wrong. It also silently assumed ~zero fixed
# setup before dispatch, which stopped holding once `_run_reference` began
# probing each reference model's context window.
#
# A barrier asserts the invariant directly: both sleeping references must
# be inside `call_llm` AT THE SAME TIME before either is allowed to return.
# If the fan-out ever runs serially the first reference blocks forever
# waiting for a partner that will not arrive, `wait()` raises BrokenBarrier
# on timeout, and the test fails with an explicit message. No wall-clock
# constant, no load sensitivity, and still a hard failure on serialization.
SLEEPERS = 2
rendezvous = threading.Barrier(SLEEPERS)
overlapped = threading.Event()
entered: list[str] = []

def slow_call_llm(**kwargs):
model = kwargs["model"]
entered.append(model)
if model == "boom":
barrier_hits.append(("enter", model, time.monotonic()))
raise RuntimeError("kaboom")
barrier_hits.append(("enter", model, time.monotonic()))
time.sleep(0.5)
barrier_hits.append(("exit", model, time.monotonic()))
# Generous relative to real scheduling latency, but finite so a serial
# regression fails fast instead of hanging the suite.
rendezvous.wait(timeout=30)
overlapped.set()
return _response(f"resp-{kwargs['provider']}")

monkeypatch.setattr(moa_loop, "call_llm", slow_call_llm)
Expand All @@ -1097,45 +1104,22 @@ def slow_call_llm(**kwargs):
{"provider": "p3", "model": "ok"},
]

start = time.monotonic()
out = moa_loop._run_references_parallel(
refs, [{"role": "user", "content": "hi"}], temperature=0.6, max_tokens=64
)
elapsed = time.monotonic() - start
try:
out = moa_loop._run_references_parallel(
refs, [{"role": "user", "content": "hi"}], temperature=0.6, max_tokens=64
)
except threading.BrokenBarrierError: # pragma: no cover - serial regression
pytest.fail(
"references did not run in parallel: a reference reached the "
"rendezvous alone, so the fan-out never had "
f"{SLEEPERS} references inside call_llm at once (entered: {entered})"
)

# Assert the INVARIANT (concurrency), not a wall-clock constant.
#
# The old form asserted `elapsed < 0.95`, i.e. "total wall time is under the
# 1.0s serial floor of two 0.5s sleeps". That silently assumed ~zero fixed
# setup before dispatch. It no longer holds: `_run_reference` now calls the
# upstream-new `_trim_messages_for_reference` (#60345 — trims the advisory
# request to the REFERENCE model's own window), whose
# `get_model_context_length` probe costs ~0.4s per distinct (provider,
# model) for models with no cached metadata (as here — "ok"/"boom" are
# fabricated). That overhead is itself incurred CONCURRENTLY in the worker
# threads, so the fan-out is still fully parallel — but `elapsed` becomes
# ~0.4 + 0.5 ≈ 0.95s+ and trips a threshold that was never measuring setup.
#
# The real contract is that the sleeping calls OVERLAP. Assert that
# directly from the timestamps the test already collects: the second
# reference must enter `call_llm` before the first one exits. This is exact,
# immune to fixed setup cost and machine load, and still fails hard if the
# references ever run serially.
sleepers = [h for h in barrier_hits if h[1] == "ok"]
enters = sorted(t for kind, _model, t in sleepers if kind == "enter")
exits = sorted(t for kind, _model, t in sleepers if kind == "exit")
assert len(enters) == 2 and len(exits) == 2, f"expected 2 sleeping refs, got {barrier_hits}"
assert enters[1] < exits[0], (
"references did not run in parallel: the second reference entered "
f"call_llm at +{enters[1] - start:.2f}s, after the first exited at "
f"+{exits[0] - start:.2f}s (total {elapsed:.2f}s)"
)
# Sanity bound: overlapping 0.5s sleeps must never approach the 1.0s serial
# floor of sleep time itself. Measured from the first sleeper's entry so
# per-reference setup (the context-window probe above) is excluded.
assert exits[-1] - enters[0] < 0.95, (
f"sleep phase took {exits[-1] - enters[0]:.2f}s — at/over the serial floor"
# The barrier only clears when both sleepers are inside call_llm together.
assert overlapped.is_set(), (
f"references never overlapped inside call_llm (entered: {entered})"
)
assert entered.count("ok") == SLEEPERS, f"expected 2 sleeping refs, got {entered}"
# Output order matches input order (stable Reference N labelling).
assert [label for label, _, _ in out] == ["p1:ok", "moa:preset", "p2:boom", "p3:ok"]
assert "recursively reference MoA" in out[1][1]
Expand Down
Loading