diff --git a/tests/e2e/core/_pending_fixes.py b/tests/e2e/core/_pending_fixes.py index 3e47761709e19..7eabc989b127f 100644 --- a/tests/e2e/core/_pending_fixes.py +++ b/tests/e2e/core/_pending_fixes.py @@ -12,7 +12,7 @@ import contextlib import re -from typing import Iterator, Tuple, Type +from typing import ContextManager, Iterator, Mapping, Tuple, Type import pytest @@ -30,3 +30,11 @@ def known_failure(pattern: str, reason: str, if not re.search(pattern, str(exc)): raise pytest.xfail(f"{reason} [observed: {str(exc).splitlines()[0][:240]}]") + + +def known_gate(known: Mapping[str, Tuple[str, str]], key: str, + raises: Type[BaseException] | Tuple[Type[BaseException], ...] = AssertionError) -> ContextManager[None]: + """:func:`known_failure` for a cell a file's ``KNOWN`` table (key -> ``(pattern, reason)``) + names, a no-op for every other cell, so one wrapped block serves gated and plain cells.""" + entry = known.get(key) + return known_failure(*entry, raises=raises) if entry else contextlib.nullcontext() diff --git a/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py b/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py index c5b6c226a0a3d..482bbd00cc019 100644 --- a/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py +++ b/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py @@ -290,9 +290,9 @@ def _assert_heartbeat(scn: Scenario, hb: Heartbeat, stats: dict[str, Any], turn_ class ToolOutlivedGateway(Exception): """The in-flight tool outlived the gateway's exit: its process tree survives and/or its - tool_call was left with no result in state.db. Deliberately NOT an AssertionError: the - known-bug xfail matches only this, so an RPC failure, a crash or any other broken invariant - (heartbeat, exit deadline, integrity) still fails the test.""" + tool_call was left with no result in state.db. Deliberately NOT an AssertionError: a + ``known_failure(..., raises=ToolOutlivedGateway)`` gate for this bug accepts only it, so an RPC + failure, a crash or any other broken invariant (heartbeat, exit deadline, integrity) still fails.""" def _exit_mid_turn(scn: Scenario, gw: TuiGatewayProcess, hb: Heartbeat, tag: str, @@ -456,12 +456,7 @@ def scenario_futures(request: pytest.FixtureRequest, tmp_path_factory: pytest.Te fut.cancel() -KNOWN_BUGS: dict = {} - - -@pytest.mark.parametrize("scn", [ - pytest.param(s, id=s.id, marks=[KNOWN_BUGS[s.id]] if s.id in KNOWN_BUGS else []) - for s in SCENARIOS]) +@pytest.mark.parametrize("scn", [pytest.param(s, id=s.id) for s in SCENARIOS]) def test_tui_gateway_turn_stays_live_under_fault(scn: Scenario, scenario_futures) -> None: stats = scenario_futures[scn.id].result(timeout=WARMUP_DEADLINE_S + 3 * TURN_DEADLINE_S + 120) print(json.dumps(stats)) diff --git a/tests/e2e/core/dashboard/_issue_helpers.py b/tests/e2e/core/dashboard/_issue_helpers.py index f3f8d2337005e..5459d4dc9cc56 100644 --- a/tests/e2e/core/dashboard/_issue_helpers.py +++ b/tests/e2e/core/dashboard/_issue_helpers.py @@ -8,8 +8,10 @@ * ``GatewayApiServer``: a real ``python -m gateway.run`` with the API-server platform enabled via the profile ``.env`` exactly as a user configures it (``API_SERVER_ENABLED``/``_KEY``/``_HOST``/ ``_PORT``), retrying the pick-then-bind port race like the parity lane's driver. -* ``KnownIssue`` assertion subclasses: a strict xfail cell declares ``raises=`` so - only the bug's own signature is excused; a boot failure or any other assertion stays red. +* ``KnownIssue`` assertion subclasses: a KNOWN cell gates its final check with + ``known_gate(KNOWN, name, raises=)`` (tests/e2e/core/_pending_fixes.py), so only the + bug's own signature is excused; a boot failure or any other assertion stays red, and the cell + simply passes once the fix lands. """ from __future__ import annotations @@ -48,14 +50,6 @@ class Issue120937(KnownIssue): """#120937: /api/sessions/{id}/chat silently truncates a long message.""" -def xfail_known(known: dict[str, tuple[str, type[KnownIssue]]], name: str): - """``pytest.mark.xfail(strict=True)`` for a KNOWN entry, excusing ONLY its issue's exception.""" - import pytest - - reason, exc = known[name] - return pytest.mark.xfail(strict=True, raises=exc, reason=reason) - - # Dashboard with a TTY stdin ----------------------------------------------------------------------- diff --git a/tests/e2e/core/dashboard/test_api_session_chat_limits.py b/tests/e2e/core/dashboard/test_api_session_chat_limits.py index 65616e9a54375..577018232a25b 100644 --- a/tests/e2e/core/dashboard/test_api_session_chat_limits.py +++ b/tests/e2e/core/dashboard/test_api_session_chat_limits.py @@ -22,15 +22,18 @@ import pytest from tests.e2e.core.dashboard._helpers import Sandbox, make_sandbox -from tests.e2e.core.dashboard._issue_helpers import GatewayApiServer, Issue120937, xfail_known +from tests.e2e.core._pending_fixes import known_gate +from tests.e2e.core.dashboard._issue_helpers import GatewayApiServer, Issue120937 pytestmark = pytest.mark.skipif(not sys.platform.startswith("linux"), reason="reaper reads /proc") TURN_TIMEOUT = 90.0 -KNOWN: dict[str, tuple[str, type]] = { +# test name -> (the bug's own failure signature, "#issue reason"); see _pending_fixes.known_failure. +KNOWN: dict[str, tuple[str, str]] = { "test_session_chat_100k_message_reaches_model_whole_or_is_rejected": ( + r"^#120937 /api/sessions/\{id\}/chat answered 200 but the model received 65536 of 100000 chars", "#120937 POST /api/sessions/{id}/chat silently truncates a string message to 65,536 chars " - "(200, no error, no flag)", Issue120937), + "(200, no error, no flag)"), } @@ -128,7 +131,6 @@ def test_session_chat_60k_and_runs_100k_reach_model_whole(gateway: tuple[Sandbox assert run.get("status") == "completed", f"/v1/runs/{run_id} -> {run}" -@xfail_known(KNOWN, "test_session_chat_100k_message_reaches_model_whole_or_is_rejected") def test_session_chat_100k_message_reaches_model_whole_or_is_rejected( gateway: tuple[Sandbox, GatewayApiServer]) -> None: sb, gw = gateway @@ -140,8 +142,9 @@ def test_session_chat_100k_message_reaches_model_whole_or_is_rejected( return assert status == 200, f"/chat 100k -> {status}: {str(body)[:500]}\n{gw.log_tail()}" seen = _model_saw(sb, head) - if len(seen) < len(msg) or tail not in seen: - raise Issue120937( - f"#120937 /api/sessions/{{id}}/chat answered 200 but the model received {len(seen)} of " - f"{len(msg)} chars (tail canary {'present' if tail in seen else 'MISSING'}; stored user " - f"message length {_stored_user_len(gw, sid, head)}; response keys {sorted(body)[:12]})") + with known_gate(KNOWN, "test_session_chat_100k_message_reaches_model_whole_or_is_rejected", raises=Issue120937): + if len(seen) < len(msg) or tail not in seen: + raise Issue120937( + f"#120937 /api/sessions/{{id}}/chat answered 200 but the model received {len(seen)} of " + f"{len(msg)} chars (tail canary {'present' if tail in seen else 'MISSING'}; stored user " + f"message length {_stored_user_len(gw, sid, head)}; response keys {sorted(body)[:12]})") diff --git a/tests/e2e/core/dashboard/test_dashboard_mcp_install.py b/tests/e2e/core/dashboard/test_dashboard_mcp_install.py index 3ce450b7725f5..e9dda1d889bae 100644 --- a/tests/e2e/core/dashboard/test_dashboard_mcp_install.py +++ b/tests/e2e/core/dashboard/test_dashboard_mcp_install.py @@ -28,19 +28,24 @@ import yaml from tests.e2e.core.dashboard._helpers import Sandbox, make_sandbox -from tests.e2e.core.dashboard._issue_helpers import Issue120527, PtyDashboard, xfail_known +from tests.e2e.core._pending_fixes import known_gate +from tests.e2e.core.dashboard._issue_helpers import Issue120527, PtyDashboard pytestmark = pytest.mark.skipif(not sys.platform.startswith("linux"), reason="pty stdin + /proc reaper") INSTALL_DEADLINE_S = 30.0 # a healthy install (write config + spawn/probe a local server) takes ~2 s FOLLOWUP_DEADLINE_S = 5.0 FIXTURE = Path(__file__).with_name("fixture_catalog_mcp.py") -# Open issues this file encodes. A strict xfail XPASSes (and fails) the moment the fix lands, so the -# entry is removed with the fix; ``raises`` excuses only the issue's own assertion class. -KNOWN: dict[str, tuple[str, type]] = { +# Open issues this file encodes: test name -> (the bug's own failure signature, "#issue reason"). The +# gate (_pending_fixes.known_failure) excuses only Issue120527 with that message and passes once the +# fix lands; delete the entry then. +KNOWN: dict[str, tuple[str, str]] = { "test_catalog_install_from_tty_launched_dashboard_never_wedges_the_server": ( + # A slow install whose follow-ups both answer is not the wedge; it must stay red. + r"^dashboard wedged by an MCP catalog install \(stdin=/dev/pts/\d+\): install -> ReadTimeout after [^;]*; " + r"(?!GET /api/config -> HTTP 200; /api/ws -> accepted\n)", "#120527 dashboard MCP catalog install reaches the interactive tool checklist on the serve " - "worker thread when stdin is a TTY and wedges _SKILLS_PROFILE_LOCK", Issue120527), + "worker thread when stdin is a TTY and wedges _SKILLS_PROFILE_LOCK"), } @@ -144,6 +149,8 @@ def test_catalog_install_from_headless_dashboard_completes_with_all_probed_tools assert block.get("enabled") is True and "tools" not in block, f"unexpected tool filter: {block}" -@xfail_known(KNOWN, "test_catalog_install_from_tty_launched_dashboard_never_wedges_the_server") def test_catalog_install_from_tty_launched_dashboard_never_wedges_the_server(sb: Sandbox, tmp_path: Path) -> None: - _install_and_probe(sb, tmp_path, tty=True, wedge_exc=Issue120527) + # Issue120527 is raised only at the wedge verdict, after the install and both follow-ups settled. + with known_gate(KNOWN, "test_catalog_install_from_tty_launched_dashboard_never_wedges_the_server", + raises=Issue120527): + _install_and_probe(sb, tmp_path, tty=True, wedge_exc=Issue120527) diff --git a/tests/e2e/core/delivery/_pending_fixes.py b/tests/e2e/core/delivery/_pending_fixes.py index 26374b9c1ec8d..d3fa443bb9c8c 100644 --- a/tests/e2e/core/delivery/_pending_fixes.py +++ b/tests/e2e/core/delivery/_pending_fixes.py @@ -3,14 +3,13 @@ A plain ``xfail(strict=True)`` turns main red the moment its fix merges (XPASS), and a non-strict one guards nothing. Instead, each gap here has a PROBE: a few lines that reproduce the defect's mechanism on the tree under test, in a throwaway interpreter with its own -``HOME``/``HERMES_HOME`` (no state leaks into the suite process). ``expect_gap`` applies a -strict xfail only while the probe still reproduces the defect; once the fix is in the tree the -cell runs as a plain test, so it must pass. Whichever lands first, suite or fix, main stays -green, and a probe that disagrees with the end-to-end cell still fails loudly (XPASS, or a -real failure) instead of hiding. +``HOME``/``HERMES_HOME`` (no state leaks into the suite process). A cell asks ``gap_open`` and +tolerates the gap only while the probe still reproduces the defect; once the fix is in the tree +the cell runs as a plain test, so it must pass. Whichever lands first, suite or fix, main stays +green, and a probe that disagrees with the end-to-end cell still fails loudly instead of hiding. Probes exercise behaviour only (never read source text). When a fix has landed, delete its -entry and every ``expect_gap`` / ``gap_open`` call naming it. +entry and every ``gap_open`` call naming it. A gap with no fix PR yet has no probe: ``known_failure`` is a run-time xfail keyed on the gap's own assertion message, so the cell XFAILs only while it fails exactly that way, fails loudly on @@ -48,40 +47,6 @@ jobs._hermes_now = lambda: datetime.fromisoformat("2026-03-07T09:00:30-05:00") nxt = jobs.compute_next_run({"kind": "cron", "expr": "0 9 * * *"}) print("fixed" if nxt == "2026-03-08T09:00:00-04:00" else "open") -'''), - # A failed first stream send disabled edits but left no message id, so the next tick sent a - # second first send: an uneditable partial preview stayed visible next to the final reply. - 120315: ({}, r''' -import asyncio -from types import SimpleNamespace -from unittest.mock import AsyncMock, MagicMock -from gateway.stream_consumer import GatewayStreamConsumer, StreamConsumerConfig - -async def main(): - delivered, results = [], iter([SimpleNamespace(success=False, error="timeout")]) - async def send(**kw): - r = next(results, None) or SimpleNamespace(success=True, message_id=f"m{len(delivered)}") - if r.success: - delivered.append(kw["content"]) - return r - adapter = MagicMock() - adapter.send = AsyncMock(side_effect=send) - adapter.edit_message = AsyncMock(return_value=SimpleNamespace(success=True)) - adapter.MAX_MESSAGE_LENGTH = 4096 - c = GatewayStreamConsumer(adapter, "chat", StreamConsumerConfig(edit_interval=0.01, - buffer_threshold=5, cursor="")) - c.on_delta("preview never landed ") - task = asyncio.create_task(c.run()) - await asyncio.sleep(0.08) - c.on_delta("and more streamed text ") - await asyncio.sleep(0.08) - c.on_delta("then the end.") - c.finish() - await asyncio.wait_for(task, timeout=10) - return delivered - -print("fixed" if asyncio.run(main()) == ["preview never landed and more streamed text then the end."] - else "open") '''), } @@ -106,13 +71,6 @@ def gap_open(pr: int) -> bool: return verdict == ["open"] -def expect_gap(request, pr: int, reason: str) -> None: - """Strict xfail for this cell while #``pr``'s defect reproduces; a plain test once it doesn't.""" - assert f"#{pr}" in reason, f"reason for a #{pr} gap must name the PR: {reason!r}" - if gap_open(pr): - request.applymarker(pytest.mark.xfail(strict=True, reason=reason)) - - @contextlib.contextmanager def known_failure(pattern: str, reason: str, on_xfail: Optional[Callable[[], None]] = None) -> Iterator[None]: diff --git a/tests/e2e/core/delivery/test_messaging_exactly_once.py b/tests/e2e/core/delivery/test_messaging_exactly_once.py index bf5cbd5713698..67274ebf9505a 100644 --- a/tests/e2e/core/delivery/test_messaging_exactly_once.py +++ b/tests/e2e/core/delivery/test_messaging_exactly_once.py @@ -46,7 +46,7 @@ visible_copies, wait_until, ) -from tests.e2e.core.delivery._pending_fixes import expect_gap, known_failure +from tests.e2e.core.delivery._pending_fixes import known_failure from tests.fakes.fake_llm_provider import FakeLLMServer, StallMidStream, Text, ToolCall pytestmark = pytest.mark.skipif(sys.platform == "win32", reason="SIGKILL + process-group restart harness") @@ -252,18 +252,6 @@ def _say(body: str, **kw): } -# Streaming gaps #120315 fixes: an overflowing first send keeps the " (n/n)" indicator on the live -# tail that later chunks extend / a sealed remainder over the limit is re-sent; a failed first send -# leaves an uneditable partial preview next to the final. Every cell here fails every run while the -# gap is open (the chunk pacing above puts each chunk in its own consumer tick). -STREAM_OVERFLOW_GAP = "LIVE GAP (fixed by #120315): streamed overflow / failed first send" -STREAM_OVERFLOW_GAP_CELLS = { - ("burst_overflow", "fk_tg"), ("burst_overflow", "fk_dc"), - ("long_streamed", "fk_dc"), - ("stream_timeout_first_send", "fk_tg"), ("stream_timeout_first_send", "fk_dc"), -} - - STREAM_ACK_LOST_GAP = ( "LIVE GAP (#53449/#25349 family): on a streaming platform the stream consumer's first send is the " "whole short reply; when the platform accepts it but the ack is lost (timeout), the consumer reports " @@ -284,7 +272,7 @@ def faults_fired(gw: GatewayProcess, aid: str) -> List[dict]: @pytest.mark.parametrize("platform", list(PLATFORMS)) @pytest.mark.parametrize("fault", list(FAULTS)) -def test_delivery_fault_matrix(gw, director, platform, fault, request): +def test_delivery_fault_matrix(gw, director, platform, fault): kind, op, target, build, expect = FAULTS[fault] token = f"{platform}.{fault}" chat = f"m-{platform}-{fault}" @@ -303,8 +291,6 @@ def test_delivery_fault_matrix(gw, director, platform, fault, request): f"{token} complete reply\n" + dump(gw, platform, chat), proc=gw.proc, log=gw.log) if kind is not None: assert faults_fired(gw, aid), f"injected {kind} never hit a platform call\n{dump(gw, platform, chat)}" - if (fault, platform) in STREAM_OVERFLOW_GAP_CELLS: - expect_gap(request, 120315, STREAM_OVERFLOW_GAP) guard = (known_failure(STREAM_ACK_LOST_SIGNATURE, STREAM_ACK_LOST_GAP, on_xfail=lambda: XFAILED_TOKENS.add(token)) if fault == "ack_lost" and STREAMING[platform] else contextlib.nullcontext()) diff --git a/tests/e2e/core/history/test_prefix_stability.py b/tests/e2e/core/history/test_prefix_stability.py index 036221150af37..ba47cbf8f4a06 100644 --- a/tests/e2e/core/history/test_prefix_stability.py +++ b/tests/e2e/core/history/test_prefix_stability.py @@ -78,12 +78,10 @@ ], } -KNOWN_BROKEN: dict[str, str] = {} - - class ToolsArrayDrift(Exception): - """The tools array changed between requests of one session. Not an AssertionError: the - known-bug xfail matches only this, so every other invariant still fails the test.""" + """The tools array changed between requests of one session. Its own type, so a future tools-drift + bug can be gated with ``known_failure(..., raises=ToolsArrayDrift)`` without excusing any other + invariant.""" @pytest.fixture @@ -166,11 +164,7 @@ def run_journey(world: dict, hops: list[Hop]) -> tuple[str, list[tuple[int, str] return sid, openings, compaction_idx -@pytest.mark.parametrize("journey", [ - pytest.param(name, marks=pytest.mark.xfail(strict=True, raises=ToolsArrayDrift, reason=KNOWN_BROKEN[name])) - if name in KNOWN_BROKEN else name - for name in JOURNEYS -]) +@pytest.mark.parametrize("journey", list(JOURNEYS)) def test_request_prefix_is_byte_stable_across_processes(world, journey): sid, openings, compaction_idx = run_journey(world, JOURNEYS[journey]) srv, home = world["srv"], world["hermes_home"] @@ -180,8 +174,8 @@ def test_request_prefix_is_byte_stable_across_processes(world, journey): def where(i: int) -> str: return f"request {i} (in {max((o for o in openings if o[0] <= i), default=(0, '?'))[1]})" - # System prompt + messages first; the tools array is checked last, on its own, so a known - # tools-drift xfail cannot mask a message-prefix, usage or integrity regression. + # System prompt + messages first; the tools array is checked last, on its own, so a gated + # tools-drift bug cannot mask a message-prefix, usage or integrity regression. breaks = prefix_breaks(main, tools=False) unexpected = [(i, why) for i, why in breaks if i != compaction_idx] assert not unexpected, "prompt-cache prefix broke outside the compaction boundary:\n" + "\n".join( diff --git a/tests/e2e/core/kanban/test_kanban_decompose_billing.py b/tests/e2e/core/kanban/test_kanban_decompose_billing.py index e87e3cceb0ace..553535821892f 100644 --- a/tests/e2e/core/kanban/test_kanban_decompose_billing.py +++ b/tests/e2e/core/kanban/test_kanban_decompose_billing.py @@ -30,6 +30,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.kanban._helpers import PY, Board, wait_until from tests.fakes.fake_llm_provider import Error, FakeLLMServer, Response, Text @@ -39,10 +40,15 @@ pytest.mark.skipif(sys.platform == "win32", reason="POSIX process groups for the gateway child"), ] -KNOWN = { - "malformed": "#118872 a triage card whose decompose reply is unusable is re-billed every tick forever", - "http_500": "#118603 a triage card whose decompose call 5xxs is retried every dispatcher tick forever", - "test_huge_decompose_reply_is_bounded": "#118607 a 500-child decompose reply creates 500 child rows", +_REBILLED = r"doomed card billed \d+ aux decompose calls over >= \d+ dispatcher ticks \(bound \d+\)" +KNOWN: dict[str, tuple[str, str]] = { + "malformed": (_REBILLED, + "#118872 a triage card whose decompose reply is unusable is re-billed every tick forever"), + "http_500": (_REBILLED, + "#118603 a triage card whose decompose call 5xxs is retried every dispatcher tick forever"), + "test_huge_decompose_reply_is_bounded": ( + r"500-child reply created \d+ child rows \(bound \d+\)", + "#118607 a 500-child decompose reply creates 500 child rows"), } # Dispatcher knobs through documented config: tick every second, never spawn workers (we only @@ -63,15 +69,11 @@ class RebilledEveryTick(AssertionError): - """The doomed card was billed on (nearly) every tick; the only failure the known xfail covers.""" + """The doomed card was billed on (nearly) every tick; the only type ``known_gate`` accepts here.""" class UnboundedFanout(AssertionError): - """A decompose reply fanned out past the child bound; the only failure the known xfail covers.""" - - -def _known(name: str, exc: type[BaseException]) -> list: - return [pytest.mark.xfail(strict=True, raises=exc, reason=KNOWN[name])] if name in KNOWN else [] + """A decompose reply fanned out past the child bound; the only type ``known_gate`` accepts here.""" # fake aux model ---------------------------------------------------------------------------------- @@ -170,8 +172,7 @@ def prove_ticks(b: Board, proc: subprocess.Popen, k: int) -> None: # tests ------------------------------------------------------------------------------------------- -@pytest.mark.parametrize("failure", [pytest.param(name, marks=_known(name, RebilledEveryTick)) - for name in FAILURES]) +@pytest.mark.parametrize("failure", list(FAILURES)) def test_failing_triage_card_is_not_rebilled_every_tick(tmp_path: Path, failure: str) -> None: model = AuxModel() with FakeLLMServer(aux=model) as srv: @@ -187,10 +188,11 @@ def test_failing_triage_card_is_not_rebilled_every_tick(tmp_path: Path, failure: # Harness invariants (stay red regardless of the known bug): the failure produced no graph. assert children_of(b, doomed) == [], b.diag(doomed) assert b.events(doomed, "decomposed") == [], b.diag(doomed) - if bills > MAX_FAILED_ATTEMPTS: - raise RebilledEveryTick( - f"doomed card billed {bills} aux decompose calls over >= {PROBE_TICKS + 1} dispatcher " - f"ticks (bound {MAX_FAILED_ATTEMPTS}); status={b.task(doomed)['status']}") + with known_gate(KNOWN, failure, raises=RebilledEveryTick): + if bills > MAX_FAILED_ATTEMPTS: + raise RebilledEveryTick( + f"doomed card billed {bills} aux decompose calls over >= {PROBE_TICKS + 1} dispatcher " + f"ticks (bound {MAX_FAILED_ATTEMPTS}); status={b.task(doomed)['status']}") def test_valid_three_child_graph_bills_once_and_links_children(tmp_path: Path) -> None: @@ -220,8 +222,6 @@ def test_valid_three_child_graph_bills_once_and_links_children(tmp_path: Path) - [rows[f"child {i}"] for i in range(3)]] -@pytest.mark.xfail(strict=True, raises=UnboundedFanout, - reason=KNOWN["test_huge_decompose_reply_is_bounded"]) def test_huge_decompose_reply_is_bounded(tmp_path: Path) -> None: """One decompose call answering 500 children must be rejected or capped, not fanned out. Driven through the real ``hermes kanban decompose`` CLI (same decompose_task path).""" @@ -236,5 +236,6 @@ def test_huge_decompose_reply_is_bounded(tmp_path: Path) -> None: # A rejection leaves the root in triage with zero children; a cap leaves <= the bound. assert kids or b.task(root)["status"] == "triage", b.diag(root) assert task_count(b) == 1 + len(kids), "child rows exist that are not linked under the root" - if len(kids) > MAX_CHILDREN: - raise UnboundedFanout(f"500-child reply created {len(kids)} child rows (bound {MAX_CHILDREN})") + with known_gate(KNOWN, "test_huge_decompose_reply_is_bounded", raises=UnboundedFanout): + if len(kids) > MAX_CHILDREN: + raise UnboundedFanout(f"500-child reply created {len(kids)} child rows (bound {MAX_CHILDREN})") diff --git a/tests/e2e/core/kanban/test_kanban_rate_limit_review.py b/tests/e2e/core/kanban/test_kanban_rate_limit_review.py index e51ec23e1a915..cced7efa5d65b 100644 --- a/tests/e2e/core/kanban/test_kanban_rate_limit_review.py +++ b/tests/e2e/core/kanban/test_kanban_rate_limit_review.py @@ -29,6 +29,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.kanban._helpers import Board from tests.fakes.fake_llm_provider import Error, FakeLLMServer, Text, ToolCall @@ -46,20 +47,17 @@ RATE_LIMIT_EXIT_CODE = 75 # KANBAN_RATE_LIMIT_EXIT_CODE — documented worker exit contract MAX_TICKS = 6 # a healthy rate-limited card is done on tick 3; the rest prove "forever" -# Scenario -> reason. A listed scenario is a strict xfail that only matches its dedicated exception. -KNOWN: dict[str, str] = { - "rate_limited_then_review": "#119070 stale rate-limit stamp parks the review handoff as blocker_auth", +# Scenario -> (pattern, reason) for ``known_gate``; it only matches ``ReviewerNeverSpawned``. +KNOWN: dict[str, tuple[str, str]] = { + "rate_limited_then_review": ( + r"reviewer attempts=0, respawn_guarded=\[[^\]]*'blocker_auth'", + "#119070 stale rate-limit stamp parks the review handoff as blocker_auth"), } class ReviewerNeverSpawned(AssertionError): - """The card reached ``review`` but the review lane never started a reviewer (#119070).""" - - -def _known(name: str): - if name not in KNOWN: - return () - return (pytest.mark.xfail(strict=True, raises=ReviewerNeverSpawned, reason=KNOWN[name]),) + """The card reached ``review`` but the review lane never started a reviewer (#119070); the only + type ``known_gate`` accepts here.""" # fake provider ----------------------------------------------------------------------------------- @@ -231,7 +229,7 @@ def test_clean_handoff_spawns_the_reviewer_on_the_next_tick(tmp_path: Path) -> N assert b.task(tid)["consecutive_failures"] == 0, diag -@pytest.mark.parametrize("scenario", [pytest.param("rate_limited_then_review", marks=_known("rate_limited_then_review"))]) +@pytest.mark.parametrize("scenario", ["rate_limited_then_review"]) def test_rate_limited_then_review_handoff_reaches_the_reviewer(rate_limited_flow: Flow, scenario: str) -> None: f = rate_limited_flow b, tid, diag = f.board, f.tid, f.diag() @@ -240,10 +238,11 @@ def test_rate_limited_then_review_handoff_reaches_the_reviewer(rate_limited_flow assert f.run_outcomes()[:2] == ["rate_limited", "review_requested"], diag reviewers = f.model.by_role("review") guarded = [g for t in f.ticks for g in t["guarded"]] - if not reviewers or b.task(tid)["status"] != "done": - raise ReviewerNeverSpawned( - f"{scenario}: status={b.task(tid)['status']} after {len(f.ticks)} ticks, " - f"reviewer attempts={len(reviewers)}, respawn_guarded={guarded}\n{diag}") + with known_gate(KNOWN, scenario, raises=ReviewerNeverSpawned): + if not reviewers or b.task(tid)["status"] != "done": + raise ReviewerNeverSpawned( + f"{scenario}: status={b.task(tid)['status']} after {len(f.ticks)} ticks, " + f"reviewer attempts={len(reviewers)}, respawn_guarded={guarded}\n{diag}") # Once fixed, the whole contract must hold, not just "something spawned". assert f.run_outcomes() == ["rate_limited", "review_requested", "completed"], diag # The reviewer starts on the tick right after the handoff (cooldown 0), billed one tool turn. diff --git a/tests/e2e/core/kanban/test_kanban_worker_contract.py b/tests/e2e/core/kanban/test_kanban_worker_contract.py index 0f429241f7afd..b6d8d1d9246d8 100644 --- a/tests/e2e/core/kanban/test_kanban_worker_contract.py +++ b/tests/e2e/core/kanban/test_kanban_worker_contract.py @@ -19,6 +19,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.kanban._helpers import Board, wait_until from tests.fakes.fake_llm_provider import FakeLLMServer, Text, ToolCall @@ -27,10 +28,12 @@ pytest.mark.live_system_guard_bypass, # teardown SIGKILLs this board's reparented workers ] -KNOWN: dict[str, str] = { - "outside": "#120647 kanban_complete artifact outside the workspace is silently never attached", - "test_worker_with_unresolvable_pinned_skill_still_starts_its_session": - "#119619 dispatcher-owned worker with a stale --skill pin exits rc=1 before its session", +KNOWN: dict[str, tuple[str, str]] = { + "outside": (r"card done with its declared outside-workspace artifact never attached", + "#120647 kanban_complete artifact outside the workspace is silently never attached"), + "test_worker_with_unresolvable_pinned_skill_still_starts_its_session": ( + r"worker died before its session \(exits=", + "#119619 dispatcher-owned worker with a stale --skill pin exits rc=1 before its session"), } _TASK_RE = re.compile(r"work kanban task (t_[0-9a-f]+)") @@ -40,7 +43,8 @@ class KnownGap(AssertionError): - """The tracked bug's own assertion. Only this is xfailed; any harness failure stays red.""" + """The tracked bug's own assertion, the only type ``known_gate`` accepts; any harness failure + stays red.""" def _task_id(rec: dict) -> str: @@ -67,10 +71,7 @@ def _artifact_location(board: Board, tid: str, where: str) -> Path: "outside": board.hermes_home / "scripts" / f"{tid}-deliverable.md"}[where] -@pytest.mark.parametrize("where", [ - pytest.param(w, marks=pytest.mark.xfail(strict=True, raises=KnownGap, reason=KNOWN[w])) if w in KNOWN else w - for w in ("inside", "outside") -]) +@pytest.mark.parametrize("where", ["inside", "outside"]) def test_declared_artifact_is_attached_or_reported(tmp_path, where: str) -> None: board_ref: dict[str, Board] = {} blocked: dict[str, bool] = {} @@ -98,9 +99,10 @@ def responder(rec: dict): status = board.task(tid)["status"] attached = board._q("SELECT * FROM task_attachments WHERE task_id = ?", (tid,)) stored = [Path(a["stored_path"]) for a in attached] - if status == "done" and not attached: - raise KnownGap(f"card done with its declared {where}-workspace artifact never attached\n" - f"{board.diag(tid)}") + with known_gate(KNOWN, where, raises=KnownGap): + if status == "done" and not attached: + raise KnownGap(f"card done with its declared {where}-workspace artifact never attached\n" + f"{board.diag(tid)}") assert status == "done" or where == "outside", board.diag(tid) if status != "done": _assert_visible_refusal(board, srv, tid, _artifact_location(board, tid, where)) @@ -161,8 +163,6 @@ def test_worker_with_resolvable_pinned_skill_sees_it_on_first_request(tmp_path) board.kill_workers() -@pytest.mark.xfail(strict=True, raises=KnownGap, - reason=KNOWN["test_worker_with_unresolvable_pinned_skill_still_starts_its_session"]) def test_worker_with_unresolvable_pinned_skill_still_starts_its_session(tmp_path) -> None: with FakeLLMServer(_completing_responder) as srv: board = Board(tmp_path, srv.base_url) @@ -176,8 +176,10 @@ def test_worker_with_unresolvable_pinned_skill_still_starts_its_session(tmp_path exits = [int(rc) for rc in _EXIT_RE.findall(log)] # The bug's own signature: no model call at all, and the worker died of the stale pin # (nonzero exit trailer, or its log names the missing skill). Anything else stays red. - if not srv.main_requests() and ((exits and exits[-1] != 0) or "e2e-archived" in log): - raise KnownGap(f"worker died before its session (exits={exits}); board:\n{board.diag(tid)}") + with known_gate(KNOWN, "test_worker_with_unresolvable_pinned_skill_still_starts_its_session", + raises=KnownGap): + if not srv.main_requests() and ((exits and exits[-1] != 0) or "e2e-archived" in log): + raise KnownGap(f"worker died before its session (exits={exits}); board:\n{board.diag(tid)}") assert SKILL_MARK not in str(srv.main_requests()[0]["messages"]), "removed skill still loaded" assert board.task(tid)["status"] == "done", board.diag(tid) finally: diff --git a/tests/e2e/core/kanban/test_kanban_worker_sigkill.py b/tests/e2e/core/kanban/test_kanban_worker_sigkill.py index 981e124b77c0a..ab5e37c7f3191 100644 --- a/tests/e2e/core/kanban/test_kanban_worker_sigkill.py +++ b/tests/e2e/core/kanban/test_kanban_worker_sigkill.py @@ -24,6 +24,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.kanban._helpers import Board, pid_alive, wait_until from tests.fakes.fake_llm_provider import FakeLLMServer, Hang, Text, ToolCall @@ -40,24 +41,24 @@ ATTEMPT_ONE_PROSE = "ATTEMPT_ONE_PROSE the card could not be finished in this run." ATTEMPT_TWO_MARK = "ATTEMPT_TWO_MARK heartbeat sent, about to call the provider again." -# Red on current main for a tracked, open bug. Strict: the test FAILS the moment the bug is fixed, -# forcing the entry out so the contract is enforced again. -KNOWN: dict[str, str] = { - "test_fresh_claim_does_not_inherit_previous_heartbeat": - "#119155 a fresh claim keeps the previous run's last_heartbeat_at", - "test_sigkilled_attempt_is_booked_as_a_crash_of_its_own": - "#121255 a SIGKILLed worker is booked with the previous attempt's rc=0 trailer", - "test_crash_diagnostic_comes_from_the_crashed_attempt": - "#119618 crash diagnostic carries an older attempt's output", +# Open tracked bugs, test name -> (pattern, reason) for ``known_gate`` (matches ``KnownGap`` only). +# Delete an entry once its fix lands. +KNOWN: dict[str, tuple[str, str]] = { + "test_fresh_claim_does_not_inherit_previous_heartbeat": ( + r"fresh run \d+ \(started \d+\) carries last_heartbeat_at=\d+, attempt 2's value", + "#119155 a fresh claim keeps the previous run's last_heartbeat_at"), + "test_sigkilled_attempt_is_booked_as_a_crash_of_its_own": ( + r"SIGKILLed attempt booked as \w+ .*'exit_code': 0\b", + "#121255 a SIGKILLed worker is booked with the previous attempt's rc=0 trailer"), + "test_crash_diagnostic_comes_from_the_crashed_attempt": ( + r"attempt 2's crash diagnostic quotes attempt 1: ", + "#119618 crash diagnostic carries an older attempt's output"), } class KnownGap(AssertionError): - """The tracked bug's own assertion. Only this is xfailed; any harness failure stays red.""" - - -def known(name: str): - return pytest.mark.xfail(strict=True, raises=KnownGap, reason=KNOWN[name]) + """The tracked bug's own assertion, the only type ``known_gate`` accepts; any harness failure + stays red.""" class Director: @@ -166,20 +167,19 @@ def test_sigkilled_worker_is_reclaimed_and_the_retry_completes_once(scenario: Sc assert len(b.events(sc.tid, "completed")) == 1 -@known("test_fresh_claim_does_not_inherit_previous_heartbeat") def test_fresh_claim_does_not_inherit_previous_heartbeat(scenario: Scenario) -> None: sc = scenario run = sc.claim_run assert run["outcome"] is None and run["status"] == "running", sc.board.diag(sc.tid) assert sc.after_claim["current_run_id"] == run["id"] hb = sc.after_claim["last_heartbeat_at"] - if hb is not None and int(hb) < int(run["started_at"]): - raise KnownGap( - f"fresh run {run['id']} (started {run['started_at']}) carries last_heartbeat_at={hb}, " - f"attempt 2's value {sc.hb_before_kill}") + with known_gate(KNOWN, "test_fresh_claim_does_not_inherit_previous_heartbeat", raises=KnownGap): + if hb is not None and int(hb) < int(run["started_at"]): + raise KnownGap( + f"fresh run {run['id']} (started {run['started_at']}) carries last_heartbeat_at={hb}, " + f"attempt 2's value {sc.hb_before_kill}") -@known("test_sigkilled_attempt_is_booked_as_a_crash_of_its_own") def test_sigkilled_attempt_is_booked_as_a_crash_of_its_own(scenario: Scenario) -> None: sc = scenario run2 = _run_of_attempt(sc, 2) @@ -188,15 +188,16 @@ def test_sigkilled_attempt_is_booked_as_a_crash_of_its_own(scenario: Scenario) - booked = [(k, p) for k, p in kinds if k in ("crashed", "protocol_violation", "rate_limited")] assert len(booked) == 1, kinds kind, payload = booked[0] - if kind != "crashed" or (payload or {}).get("exit_code") == 0: - raise KnownGap(f"SIGKILLed attempt booked as {kind} {payload}") + with known_gate(KNOWN, "test_sigkilled_attempt_is_booked_as_a_crash_of_its_own", raises=KnownGap): + if kind != "crashed" or (payload or {}).get("exit_code") == 0: + raise KnownGap(f"SIGKILLed attempt booked as {kind} {payload}") -@known("test_crash_diagnostic_comes_from_the_crashed_attempt") def test_crash_diagnostic_comes_from_the_crashed_attempt(scenario: Scenario) -> None: sc = scenario run1, run2 = _run_of_attempt(sc, 1), _run_of_attempt(sc, 2) # Vacuity guard: attempt 1's own diagnostic does carry its prose. assert "ATTEMPT_ONE_PROSE" in (run1["error"] or ""), run1 - if "ATTEMPT_ONE_PROSE" in (run2["error"] or ""): - raise KnownGap(f"attempt 2's crash diagnostic quotes attempt 1: {run2['error'][-300:]!r}") + with known_gate(KNOWN, "test_crash_diagnostic_comes_from_the_crashed_attempt", raises=KnownGap): + if "ATTEMPT_ONE_PROSE" in (run2["error"] or ""): + raise KnownGap(f"attempt 2's crash diagnostic quotes attempt 1: {run2['error'][-300:]!r}") diff --git a/tests/e2e/core/mcp_plugins/_helpers.py b/tests/e2e/core/mcp_plugins/_helpers.py index 887c3bcb0896a..b259bf728a380 100644 --- a/tests/e2e/core/mcp_plugins/_helpers.py +++ b/tests/e2e/core/mcp_plugins/_helpers.py @@ -37,7 +37,7 @@ _SECRET_ENV_SUFFIXES = ("_API_KEY", "_TOKEN", "_SECRET", "_ACCESS_KEY") _PASSTHROUGH_ENV = frozenset({"PATH", "LANG", "LANGUAGE", "USER", "LOGNAME", "SHELL", "TMPDIR", "TZ"}) -__all__ = ["FINAL", "E2EHome", "HttpMcpServer", "KnownSymptom", "apply_known", "build_home", "stdio_server", +__all__ = ["FINAL", "E2EHome", "HttpMcpServer", "KnownSymptom", "build_home", "stdio_server", "http_server_cfg", "script", "call_tool", "run_chat_q", "inbound", "calls_received", "tool_results", "tool_name", "tool_names", "payload", "symptom", "kill_tagged", "tagged_pids", "wait_until"] @@ -50,9 +50,10 @@ def tool_name(server: str, tool: str) -> str: class KnownSymptom(Exception): - """Raised ONLY by the assertion that observes a KNOWN bug's symptom. KNOWN cells are strict - xfails with ``raises=KnownSymptom``, so every other failure in them (a server that never starts, - a timeout, a precondition, a crashed host, teardown) is a plain error and stays red.""" + """Raised ONLY by :func:`symptom`, the assertion that observes a KNOWN bug's symptom. It is the + type ``known_gate(KNOWN, request.node.name, raises=KnownSymptom)`` accepts around that assertion, + so every other failure in a KNOWN cell (a server that never starts, a timeout, a precondition, a + crashed host, teardown) is a plain error and stays red.""" def symptom(ok: Any, message: str) -> None: @@ -61,13 +62,6 @@ def symptom(ok: Any, message: str) -> None: raise KnownSymptom(message) -def apply_known(request: Any, known: dict[str, str]) -> None: - """From an autouse fixture: strict-xfail the current test when its node name is a KNOWN key.""" - reason = known.get(request.node.name) - if reason: - request.applymarker(pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=reason)) - - def payload(result: str) -> dict[str, Any]: """The JSON object inside a tool result's untrusted-content wrapper.""" match = re.search(r"^\{.*\}$", result, re.M | re.S) diff --git a/tests/e2e/core/mcp_plugins/test_mcp_stdio_conformance.py b/tests/e2e/core/mcp_plugins/test_mcp_stdio_conformance.py index cc6cd585b3bc7..8d540b45ddab2 100644 --- a/tests/e2e/core/mcp_plugins/test_mcp_stdio_conformance.py +++ b/tests/e2e/core/mcp_plugins/test_mcp_stdio_conformance.py @@ -31,7 +31,7 @@ from tests.e2e.core.mcp_plugins._helpers import ( FINAL, - apply_known, + KnownSymptom, build_home, calls_received, payload, @@ -45,6 +45,7 @@ tool_results, ) from tests.e2e.core.mcp_plugins._plugin_helpers import reap_tagged +from tests.e2e.core._pending_fixes import known_gate pytestmark = [ pytest.mark.skipif(not sys.platform.startswith("linux"), reason="orphan sweep uses /proc"), @@ -54,23 +55,23 @@ SERVER = "e2e" UNSUPPORTED_IMAGES = ("svg", "avif", "tiff", "heic", "badb64") -# Open bugs on origin/main. Strict: the test FAILS the moment the bug is fixed, forcing the entry out. -KNOWN: dict[str, str] = { - "test_untrusted_server_runs_read_only_tool_without_approval": - "#121042 readOnlyHint read by camelCase attribute under mcp 2.x; every tool needs approval", - **{f"test_uncacheable_image_is_reported_to_the_model[{fmt}]": - "#120227 an MCP image the cache cannot store vanishes from the tool result" for fmt in UNSUPPORTED_IMAGES}, - **{f"test_no_required_param_tools_receive_an_arguments_object[{bridge}]": - "#120923 every tools/call carries an information-free params._meta: {} (stdio too)" +# Open bugs on origin/main: test id -> (the symptom's own message pattern, "#issue reason"). Run-time +# gated around the symptom() check only (known_gate); drop an entry when its fix lands. +KNOWN: dict[str, tuple[str, str]] = { + "test_untrusted_server_runs_read_only_tool_without_approval": ( + r"^readOnlyHint=true tool was gated on an untrusted server: server got \[\], " + r"model got .*write-capable MCP tool 'ro_probe'", + "#121042 readOnlyHint read by camelCase attribute under mcp 2.x; every tool needs approval"), + **{f"test_uncacheable_image_is_reported_to_the_model[{fmt}]": ( + rf"^{fmt} image block vanished: the model got no sign the tool returned an image", + "#120227 an MCP image the cache cannot store vanishes from the tool result") for fmt in UNSUPPORTED_IMAGES}, + **{f"test_no_required_param_tools_receive_an_arguments_object[{bridge}]": ( + r"^tools/call carried an information-free params\._meta: \[\{.*'_meta': \{\}", + "#120923 every tools/call carries an information-free params._meta: {} (stdio too)") for bridge in ("direct", "tool_call bridge")}, } -@pytest.fixture(autouse=True) -def _known(request: pytest.FixtureRequest) -> None: - apply_known(request, KNOWN) - - def _run(root: Path, calls: list[tuple[str, dict[str, Any] | str]], *, extra: dict | None = None, **server_cfg: Any) -> dict[str, Any]: """One ``chat -q`` turn issuing ``calls`` (bare MCP tool names); returns the observations.""" @@ -106,11 +107,13 @@ def test_untrusted_server_refuses_destructive_tool_before_the_rpc(untrusted: dic assert "error" in rw and untrusted["canary"] not in json.dumps(rw), rw -def test_untrusted_server_runs_read_only_tool_without_approval(untrusted: dict[str, Any]) -> None: +def test_untrusted_server_runs_read_only_tool_without_approval(untrusted: dict[str, Any], + request: pytest.FixtureRequest) -> None: received = calls_received(untrusted["log"], "ro_probe") ro = payload(untrusted["results"][0]) - symptom(received and f"RO:{untrusted['canary']}:r" in json.dumps(ro), - f"readOnlyHint=true tool was gated on an untrusted server: server got {received}, model got {ro}") + with known_gate(KNOWN, request.node.name, raises=KnownSymptom): + symptom(received and f"RO:{untrusted['canary']}:r" in json.dumps(ro), + f"readOnlyHint=true tool was gated on an untrusted server: server got {received}, model got {ro}") def test_default_trust_server_runs_destructive_tool(tmp_path: Path) -> None: @@ -132,7 +135,8 @@ def test_default_trust_server_runs_destructive_tool(tmp_path: Path) -> None: @pytest.mark.parametrize("bridge", ["direct", "tool_call bridge"]) -def test_no_required_param_tools_receive_an_arguments_object(tmp_path: Path, bridge: str) -> None: +def test_no_required_param_tools_receive_an_arguments_object(tmp_path: Path, bridge: str, + request: pytest.FixtureRequest) -> None: search = "on" if bridge == "tool_call bridge" else "off" calls = [(name, args) for name, args, _ in NO_ARG_SPELLINGS] obs = _run(tmp_path, calls, extra={"tools": {"tool_search": {"enabled": search}}}) @@ -145,7 +149,8 @@ def test_no_required_param_tools_receive_an_arguments_object(tmp_path: Path, bri assert all(obs["canary"] in r for r in obs["results"]), obs["results"] # Last, so the KNOWN symptom below can never mask a wrong-arguments failure above. empty_meta = [p for p in received if "_meta" in p and not p["_meta"]] - symptom(not empty_meta, f"tools/call carried an information-free params._meta: {empty_meta}") + with known_gate(KNOWN, request.node.name, raises=KnownSymptom): + symptom(not empty_meta, f"tools/call carried an information-free params._meta: {empty_meta}") # Image results ----------------------------------------------------------------------------------------- @@ -169,11 +174,13 @@ def test_cacheable_image_reaches_the_model_as_a_media_file(images: dict[str, Any @pytest.mark.parametrize("fmt", UNSUPPORTED_IMAGES) -def test_uncacheable_image_is_reported_to_the_model(images: dict[str, Any], fmt: str) -> None: +def test_uncacheable_image_is_reported_to_the_model(images: dict[str, Any], fmt: str, + request: pytest.FixtureRequest) -> None: block = images["by_fmt"][fmt] text = json.dumps(block) status = f"IMG-STATUS:{images['canary']}:{fmt}" assert status in text, f"the text block next to the image was lost: {block}" rest = text.replace(status, "") - symptom("image" in rest.lower(), - f"{fmt} image block vanished: the model got no sign the tool returned an image: {block}") + with known_gate(KNOWN, request.node.name, raises=KnownSymptom): + symptom("image" in rest.lower(), + f"{fmt} image block vanished: the model got no sign the tool returned an image: {block}") diff --git a/tests/e2e/core/mcp_plugins/test_mcp_streamable_http.py b/tests/e2e/core/mcp_plugins/test_mcp_streamable_http.py index c9112a15d8827..34ad74ab50fc5 100644 --- a/tests/e2e/core/mcp_plugins/test_mcp_streamable_http.py +++ b/tests/e2e/core/mcp_plugins/test_mcp_streamable_http.py @@ -27,7 +27,7 @@ from tests.e2e.core.mcp_plugins._helpers import ( FINAL, HttpMcpServer, - apply_known, + KnownSymptom, build_home, call_tool, calls_received, @@ -42,6 +42,7 @@ tool_results, ) from tests.e2e.core.mcp_plugins._plugin_helpers import reap_tagged, tui_host +from tests.e2e.core._pending_fixes import known_gate from tests.fakes.fake_llm_provider import Text pytestmark = [ @@ -51,19 +52,19 @@ SERVER = "web" -KNOWN: dict[str, str] = { - "test_no_information_free_meta_is_sent_over_http": - "#120923 empty params._meta sent on every request; some hosted MCP servers answer HTTP 400", - "test_401_on_tools_call_is_reported_as_an_auth_failure": - "#121285 mcp 2.x folds a tools/call 401 into a generic MCPError; auth recovery never runs", +# Open bugs on origin/main: test id -> (the symptom's own message pattern, "#issue reason"). Run-time +# gated around the symptom() check only (known_gate); drop an entry when its fix lands. +KNOWN: dict[str, tuple[str, str]] = { + "test_no_information_free_meta_is_sent_over_http": ( + r"^requests carried an empty/null params\._meta: \[.*'tools/call'", + "#120923 empty params._meta sent on every request; some hosted MCP servers answer HTTP 400"), + "test_401_on_tools_call_is_reported_as_an_auth_failure": ( + r"^a 401 on tools/call reached the model without any sign it is an auth failure: " + r"\{'error': 'MCP call failed: MCPError", + "#121285 mcp 2.x folds a tools/call 401 into a generic MCPError; auth recovery never runs"), } -@pytest.fixture(autouse=True) -def _known(request: pytest.FixtureRequest) -> None: - apply_known(request, KNOWN) - - def _one_turn(root: Path, calls: list[tuple[str, dict]], *, unauthorized_calls: int = 0) -> dict[str, Any]: fault = root / "401_budget" fault.write_text(str(unauthorized_calls), encoding="utf-8") @@ -96,11 +97,12 @@ def test_no_required_param_call_over_http_sends_an_arguments_object(shape: dict[ assert all(shape["canary"] in r for r in shape["results"]), shape["results"] -def test_no_information_free_meta_is_sent_over_http(shape: dict[str, Any]) -> None: +def test_no_information_free_meta_is_sent_over_http(shape: dict[str, Any], request: pytest.FixtureRequest) -> None: requests = [m for m in inbound(shape["log"]) if isinstance(m, dict) and "id" in m and "method" in m] assert requests, "the server logged no requests" empty = [m["method"] for m in requests if "_meta" in (m.get("params") or {}) and not m["params"]["_meta"]] - symptom(not empty, f"requests carried an empty/null params._meta: {empty}") + with known_gate(KNOWN, request.node.name, raises=KnownSymptom): + symptom(not empty, f"requests carried an empty/null params._meta: {empty}") # 401 on tools/call --------------------------------------------------------------------------------- @@ -112,13 +114,15 @@ def unauthorized(tmp_path_factory: pytest.TempPathFactory) -> dict[str, Any]: unauthorized_calls=1) -def test_401_on_tools_call_is_reported_as_an_auth_failure(unauthorized: dict[str, Any]) -> None: +def test_401_on_tools_call_is_reported_as_an_auth_failure(unauthorized: dict[str, Any], + request: pytest.FixtureRequest) -> None: assert any("injected_401_for" in m for m in inbound(unauthorized["log"]) if isinstance(m, dict)), ( "fixture never answered 401 (vacuous)") first = payload(unauthorized["results"][0]) assert "error" in first, first - symptom(re.search(r"auth|401|sign.?in|credential", json.dumps(first), re.I), - f"a 401 on tools/call reached the model without any sign it is an auth failure: {first}") + with known_gate(KNOWN, request.node.name, raises=KnownSymptom): + symptom(re.search(r"auth|401|sign.?in|credential", json.dumps(first), re.I), + f"a 401 on tools/call reached the model without any sign it is an auth failure: {first}") def test_the_call_after_a_401_reaches_the_server(unauthorized: dict[str, Any]) -> None: @@ -141,6 +145,8 @@ def test_the_call_after_a_401_reaches_the_server(unauthorized: dict[str, Any]) - "read-only": (["ro_probe"], 0, "RO:CANARY-crash:"), } KNOWN["test_server_crash_mid_call_fails_that_call_and_the_next_turn_reconnects[read-only]"] = ( + r"^read-only: next turn did not reach the restarted server: .*expired while this write-capable call " + r"was in flight.*NOT automatically retried.*The connection has been re-established", "#121042 readOnlyHint unseen under mcp 2.x, so a read-only call is not replayed after session expiry") @@ -159,7 +165,8 @@ def respond(record: dict[str, Any]): @pytest.mark.parametrize("kind", list(RECONNECT_PLANS)) -def test_server_crash_mid_call_fails_that_call_and_the_next_turn_reconnects(tmp_path: Path, kind: str) -> None: +def test_server_crash_mid_call_fails_that_call_and_the_next_turn_reconnects(tmp_path: Path, kind: str, + request: pytest.FixtureRequest) -> None: turn2, must_succeed, canary = RECONNECT_PLANS[kind] with provider(_planner(turn2)) as srv: eh = build_home(tmp_path, srv.base_url) @@ -181,7 +188,9 @@ def test_server_crash_mid_call_fails_that_call_and_the_next_turn_reconnects(tmp_ assert FINAL in host.turn(sid, "Use the web tool again.") after = tool_results(srv)[1:] assert len(after) == len(turn2), after - symptom(canary in after[must_succeed], f"{kind}: next turn did not reach the restarted server: {after}") + with known_gate(KNOWN, request.node.name, raises=KnownSymptom): + symptom(canary in after[must_succeed], + f"{kind}: next turn did not reach the restarted server: {after}") served_by = {m["pid"] for m in _raw(http.log) if m["msg"].get("method") == "tools/call" and (m["msg"].get("params") or {}).get("name") == turn2[0]} assert served_by and old_pid not in served_by, served_by diff --git a/tests/e2e/core/mcp_plugins/test_plugin_activation.py b/tests/e2e/core/mcp_plugins/test_plugin_activation.py index 13a634bb14a3b..6735a36d4fd23 100644 --- a/tests/e2e/core/mcp_plugins/test_plugin_activation.py +++ b/tests/e2e/core/mcp_plugins/test_plugin_activation.py @@ -16,9 +16,9 @@ ``.bak-*`` copy next to it) is what ``hermes plugins list`` shows and what a real turn runs, and the collision is reported (#121078). -Open bugs are strict xfails through ``KNOWN`` (drop the entry when its fix lands); only a -``KnownSymptom`` raised by the bug's own assertion counts as the bug, so a boot failure, a -precondition, a timeout or a crash stays red. +Open bugs are run-time gated through ``KNOWN`` (``known_gate`` around the bug's own assertion; drop +the entry when its fix lands); only a ``KnownSymptom`` raised by that assertion whose message matches +the entry's pattern counts as the bug, so a boot failure, a precondition, a timeout or a crash stays red. """ from __future__ import annotations @@ -34,7 +34,7 @@ from tests.e2e.core.mcp_plugins._helpers import ( FINAL, - apply_known, + KnownSymptom, build_home, calls_received, inbound, @@ -49,32 +49,36 @@ ) from tests.e2e.core.mcp_plugins._plugin_helpers import portable_stdio, reap_tagged, tui_host, write_portable_plugin from tests.e2e.core.parity._helpers import hermes_argv +from tests.e2e.core._pending_fixes import known_gate pytestmark = [ pytest.mark.skipif(not sys.platform.startswith("linux"), reason="process-tree cleanup uses /proc"), pytest.mark.live_system_guard_bypass, # teardown SIGKILLs only processes carrying this test's tag ] -# Confirmed-live open bugs: test id -> "#issue reason". A strict xfail XPASSes once the fix lands. -KNOWN: dict[str, str] = { - "test_resource_only_plugin_activated_live_is_reported_connected": - "#119751 live activation filters resource/prompt wrappers before the connected check", - "test_portable_mcp_env_placeholder_is_interpolated": - "#120526 portable mcp.json env ${VAR} reaches the server literally", - "test_enabled_portable_plugin_server_is_not_reported_as_an_unknown_toolset": - "#119457 startup 'Unknown toolsets' warning names a plugin-provided MCP server", - "test_same_name_backup_dir_does_not_shadow_the_live_plugin": - "#121078 the later-sorting plugins/foo.bak-* wins a same-name user collision", - "test_same_name_plugin_collision_is_reported": - "#121078 same-source manifest name collision is silent", +# Confirmed-live open bugs: test id -> (the symptom's own message pattern, "#issue reason"). Run-time +# gated around the symptom() check only (known_gate); drop an entry when its fix lands. +KNOWN: dict[str, tuple[str, str]] = { + "test_resource_only_plugin_activated_live_is_reported_connected": ( + r"^a resource-only MCP server that completed the handshake is reported as failed: " + r"\{.*'connected': False", + "#119751 live activation filters resource/prompt wrappers before the connected check"), + "test_portable_mcp_env_placeholder_is_interpolated": ( + r"^the portable plugin's server received the literal placeholder: ENV:NO-CANARY:\$\{E2E_PORTABLE_KEY\}", + "#120526 portable mcp.json env ${VAR} reaches the server literally"), + "test_enabled_portable_plugin_server_is_not_reported_as_an_unknown_toolset": ( + r"^`hermes chat` warned about the enabled plugin's MCP server: \[.*Unknown toolsets: .*\bplug\b", + "#119457 startup 'Unknown toolsets' warning names a plugin-provided MCP server"), + "test_same_name_backup_dir_does_not_shadow_the_live_plugin": ( + r"^(`hermes plugins list` shows the backup copy instead of plugins/foo \(v2\.0\.0\)" + r"|a real turn ran the backup dir's MCP server, not plugins/foo's)", + "#121078 the later-sorting plugins/foo.bak-* wins a same-name user collision"), + "test_same_name_plugin_collision_is_reported": ( + r"^two user plugin dirs declare the same name 'foo' \(.+\) but no user-visible surface names both", + "#121078 same-source manifest name collision is silent"), } -@pytest.fixture(autouse=True) -def _known(request: pytest.FixtureRequest) -> None: - apply_known(request, KNOWN) - - PLUGIN = "e2eplug" SERVER = "plug" PLUG_TOOL = tool_name(SERVER, "ro_probe") @@ -164,7 +168,7 @@ def test_plugin_enabled_mid_chat_is_usable_in_that_chat_with_an_unchanged_tools_ # 2. #119751 resource-only portable server ----------------------------------------------------- -def test_resource_only_plugin_activated_live_is_reported_connected(tmp_path: Path) -> None: +def test_resource_only_plugin_activated_live_is_reported_connected(tmp_path: Path, request: pytest.FixtureRequest) -> None: log = tmp_path / "res_inbound.jsonl" with provider(script()) as srv: eh = build_home(tmp_path, srv.base_url) @@ -176,8 +180,9 @@ def test_resource_only_plugin_activated_live_is_reported_connected(tmp_path: Pat # Guard: the server really completed the MCP handshake with this host. assert "initialize" in methods and "notifications/initialized" in methods, methods assert [r["name"] for r in rows] == [SERVER], rows - symptom(rows[0]["connected"] is True and not rows[0].get("error"), - f"a resource-only MCP server that completed the handshake is reported as failed: {rows[0]}") + with known_gate(KNOWN, request.node.name, raises=KnownSymptom): + symptom(rows[0]["connected"] is True and not rows[0].get("error"), + f"a resource-only MCP server that completed the handshake is reported as failed: {rows[0]}") # 3. #120526 ${VAR} in a portable mcp.json env (native config.yaml is the control) --------------- @@ -225,18 +230,22 @@ def test_native_mcp_env_placeholder_is_interpolated(env_echo_results: dict[str, f"native mcp_servers env ${{VAR}} did not reach the server with the .env value: {echoed}") -def test_portable_mcp_env_placeholder_is_interpolated(env_echo_results: dict[str, str]) -> None: +def test_portable_mcp_env_placeholder_is_interpolated(env_echo_results: dict[str, str], + request: pytest.FixtureRequest) -> None: echoed = _env_line(env_echo_results[SERVER]) - symptom("${E2E_PORTABLE_KEY}" not in echoed, - f"the portable plugin's server received the literal placeholder: {echoed}") + with known_gate(KNOWN, request.node.name, raises=KnownSymptom): + symptom("${E2E_PORTABLE_KEY}" not in echoed, + f"the portable plugin's server received the literal placeholder: {echoed}") assert echoed == "ENV:NO-CANARY:portable-dotenv-value", echoed -def test_enabled_portable_plugin_server_is_not_reported_as_an_unknown_toolset(env_echo_results: dict[str, str]) -> None: +def test_enabled_portable_plugin_server_is_not_reported_as_an_unknown_toolset(env_echo_results: dict[str, str], + request: pytest.FixtureRequest) -> None: """The enabled plugin's server worked in that very run (the fixture asserts its tool result), so a startup warning calling it an unknown toolset is a false alarm the user sees on every launch.""" warned = [line for line in env_echo_results["output"].splitlines() if "Unknown toolsets" in line] - symptom(not warned, f"`hermes chat` warned about the enabled plugin's MCP server: {warned}") + with known_gate(KNOWN, request.node.name, raises=KnownSymptom): + symptom(not warned, f"`hermes chat` warned about the enabled plugin's MCP server: {warned}") # 4. #121078 same manifest name in two user plugin dirs ------------------------------------------ @@ -277,17 +286,21 @@ def test_same_name_collision_still_loads_exactly_one_copy(name_collision: dict[s assert len(hits) == 1, name_collision["results"] -def test_same_name_backup_dir_does_not_shadow_the_live_plugin(name_collision: dict[str, Any]) -> None: +def test_same_name_backup_dir_does_not_shadow_the_live_plugin(name_collision: dict[str, Any], + request: pytest.FixtureRequest) -> None: rows = [line for line in name_collision["listing"].splitlines() if re.search(r"\bfoo\s*$", line)] assert rows, name_collision["listing"] - symptom("2.0.0" in rows[0], f"`hermes plugins list` shows the backup copy instead of plugins/foo (v2.0.0): {rows}") - symptom(any("RO:CANARY-LIVE:dup" in r for r in name_collision["results"]), - f"a real turn ran the backup dir's MCP server, not plugins/foo's: {name_collision['results']}") + with known_gate(KNOWN, request.node.name, raises=KnownSymptom): + symptom("2.0.0" in rows[0], + f"`hermes plugins list` shows the backup copy instead of plugins/foo (v2.0.0): {rows}") + symptom(any("RO:CANARY-LIVE:dup" in r for r in name_collision["results"]), + f"a real turn ran the backup dir's MCP server, not plugins/foo's: {name_collision['results']}") -def test_same_name_plugin_collision_is_reported(name_collision: dict[str, Any]) -> None: +def test_same_name_plugin_collision_is_reported(name_collision: dict[str, Any], request: pytest.FixtureRequest) -> None: live, backup = str(name_collision["live"]), str(name_collision["backup"]) surfaces = {"hermes plugins list": name_collision["listing"], "logs/*.log": name_collision["logs"]} named_both = [where for where, text in surfaces.items() if live in text and backup in text] - symptom(named_both, f"two user plugin dirs declare the same name 'foo' ({live} and {backup}) but no " - f"user-visible surface names both: {', '.join(surfaces)}") + with known_gate(KNOWN, request.node.name, raises=KnownSymptom): + symptom(named_both, f"two user plugin dirs declare the same name 'foo' ({live} and {backup}) but no " + f"user-visible surface names both: {', '.join(surfaces)}") diff --git a/tests/e2e/core/parity/test_entrypoint_parity.py b/tests/e2e/core/parity/test_entrypoint_parity.py index 7bc0864786a25..f0cdb077479ca 100644 --- a/tests/e2e/core/parity/test_entrypoint_parity.py +++ b/tests/e2e/core/parity/test_entrypoint_parity.py @@ -76,12 +76,6 @@ "cron (run-now)": _drive_cron.drive_cron, } -# Cells that are red on current main for a tracked, open bug. Strict: the test -# FAILS as soon as the cell turns green, so the entry is removed with the fix -# instead of silently masking a later regression of the same cell. -KNOWN_RED: dict[tuple[str, str], str] = {} - - @dataclass class Row: entrypoint: str @@ -154,10 +148,7 @@ def matrix(tmp_path_factory: pytest.TempPathFactory) -> dict[str, Row]: if out: with open(out, "a", encoding="utf-8") as fh: for row in rows.values(): - fh.write(json.dumps({ - **asdict(row), "detail": None, - "known_red": {c: ref for (ep, c), ref in KNOWN_RED.items() if ep == row.entrypoint}, - }) + "\n") + fh.write(json.dumps({**asdict(row), "detail": None}) + "\n") return rows @@ -165,10 +156,5 @@ def matrix(tmp_path_factory: pytest.TempPathFactory) -> dict[str, Row]: def test_entrypoint_parity(entrypoint: str, matrix: dict[str, Row]) -> None: row = matrix[entrypoint] assert row.error is None, f"{entrypoint}: turn failed before the cells could be evaluated:\n{row.error}" - known = {cell for (ep, cell) in KNOWN_RED if ep == entrypoint} - fixed = sorted(cell for cell in known if row.cells.get(cell)) - assert not fixed, ( - f"{entrypoint}: {fixed} now green — drop the KNOWN_RED entry " - f"({[KNOWN_RED[(entrypoint, c)] for c in fixed]}) so the cell is enforced again") - failed = sorted(k for k, ok in row.cells.items() if not ok and k not in known) + failed = sorted(k for k, ok in row.cells.items() if not ok) assert not failed, f"{entrypoint}: parity cells red: {failed}\n{row.detail}" diff --git a/tests/e2e/core/providers/_native_helpers.py b/tests/e2e/core/providers/_native_helpers.py index c6fa61ccad042..7793ccd21e087 100644 --- a/tests/e2e/core/providers/_native_helpers.py +++ b/tests/e2e/core/providers/_native_helpers.py @@ -172,8 +172,9 @@ def tool_calls_of(row: dict[str, Any]) -> list[dict[str, Any]]: class KnownSymptom(AssertionError): """Raised ONLY at a tracked bug's exact symptom. - Every KNOWN strict xfail uses ``raises=KnownSymptom`` so a harness failure (process death, timeout, - precondition assert, fixture teardown error) fails for real instead of counting as the known bug. + It is the type ``known_gate``/``known_failure`` accept (``raises=KnownSymptom``), so a harness failure + (process death, timeout, precondition assert, fixture teardown error) fails for real instead of + counting as the known bug. """ diff --git a/tests/e2e/core/providers/_openai_helpers.py b/tests/e2e/core/providers/_openai_helpers.py index d9061f7a99e08..3435c5e8de294 100644 --- a/tests/e2e/core/providers/_openai_helpers.py +++ b/tests/e2e/core/providers/_openai_helpers.py @@ -18,11 +18,12 @@ from contextlib import contextmanager from dataclasses import dataclass from pathlib import Path -from typing import Any, Callable, Iterator +from typing import Any, Callable, Iterator, Mapping -import pytest import yaml +from tests.e2e.core._pending_fixes import known_gate + REPO_ROOT = Path(__file__).resolve().parents[4] TURN_TIMEOUT = 150.0 _SECRET_ENV_SUFFIXES = ("_API_KEY", "_TOKEN", "_SECRET", "_ACCESS_KEY") @@ -34,32 +35,28 @@ class HarnessError(RuntimeError): """The harness broke (a process died, a reply or event never came, a turn overran its - budget). Never an ``AssertionError``, so a KNOWN strict xfail can never excuse it.""" + budget). Never an ``AssertionError``, so a KNOWN entry can never excuse it.""" class KnownBugError(AssertionError): - """Raised ONLY from inside ``bug_assertions()``: the one exception a KNOWN strict xfail - accepts. Preconditions, waits and teardown outside that block fail the test for real.""" - - -def known_marks(known: dict[str, str], name: str) -> list: - """Marks for a scenario: a strict xfail while ``name`` is in ``known`` (red the moment - the bug is fixed, forcing the entry out), nothing once it is gone.""" - if name not in known: - return [] - return [pytest.mark.xfail(strict=True, raises=KnownBugError, reason=known[name])] + """Raised ONLY from inside ``bug_assertions()``: the one exception type its + ``known_gate`` accepts. Preconditions, waits and teardown outside that block fail the + test for real.""" @contextmanager -def bug_assertions() -> Iterator[None]: +def bug_assertions(known: Mapping[str, tuple[str, str]], name: str) -> Iterator[None]: """Wrap ONLY the final behavioural assertions that name the bug, after every wait has - settled; an ``AssertionError`` raised inside becomes a ``KnownBugError``.""" - try: - yield - except KnownBugError: - raise - except AssertionError as exc: - raise KnownBugError(str(exc)) from exc + settled; an ``AssertionError`` raised inside becomes a ``KnownBugError``. While ``name`` + is in ``known`` (key -> ``(pattern, "#issue reason")``) a ``KnownBugError`` matching its + pattern XFAILs the cell (``known_gate``); any other failure, or a clean pass, stands.""" + with known_gate(known, name, raises=KnownBugError): + try: + yield + except KnownBugError: + raise + except AssertionError as exc: + raise KnownBugError(str(exc)) from exc # Vendor-boundary sitecustomize shims --------------------------------------------------- diff --git a/tests/e2e/core/providers/test_anthropic_discovery_and_aliases.py b/tests/e2e/core/providers/test_anthropic_discovery_and_aliases.py index 11e74723d2731..4dd83effbc858 100644 --- a/tests/e2e/core/providers/test_anthropic_discovery_and_aliases.py +++ b/tests/e2e/core/providers/test_anthropic_discovery_and_aliases.py @@ -19,6 +19,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.providers._anthropic_helpers import Rig, blocks, start_rig from tests.fakes.providers.anthropic_messages import AnthropicMessagesServer, Reply, Text, ToolUse @@ -26,29 +27,27 @@ pytest.mark.live_system_guard_bypass] class DiscoveryIgnoredEndpoint(AssertionError): - """#120844's signature: raised ONLY when the configured endpoint's catalog is not what the picker serves.""" + """#120844's signature: raised ONLY when the configured endpoint's catalog is not what the picker + serves; the type ``known_gate`` accepts for the discovery cells.""" class AliasNotReverseMapped(AssertionError): - """#120858's signature: raised ONLY when tool_describe fails to resolve the advertised wire alias.""" + """#120858's signature: raised ONLY when tool_describe fails to resolve the advertised wire alias; + the type ``known_gate`` accepts for the alias cell.""" -# Red on main for a tracked, open bug. Strict: XPASS fails, forcing the entry out with the fix; each -# xfail accepts only its dedicated exception, so harness failures in a KNOWN cell stay failures. -KNOWN: dict[str, tuple[str, type[AssertionError]]] = { - "discovery_env": ("#120844 Anthropic model discovery ignores ANTHROPIC_BASE_URL", DiscoveryIgnoredEndpoint), - "alias_describe": ("#120858 wire alias chat_history_lookup not reverse-mapped in tool_describe args", - AliasNotReverseMapped), +# Red on main for a tracked, open bug: key -> (the bug's own failure-message pattern, reason). Each +# gate accepts only its dedicated exception, so harness failures in a KNOWN cell stay failures. +KNOWN: dict[str, tuple[str, str]] = { + "discovery_env": (r"configured endpoint probed=False \(native host saw \d+ GETs\); relay catalog served=False", + "#120844 Anthropic model discovery ignores ANTHROPIC_BASE_URL"), + "alias_describe": (r"tool_describe did not resolve the advertised alias: .*not_found.*chat_history_lookup", + "#120858 wire alias chat_history_lookup not reverse-mapped in tool_describe args"), } RELAY_ONLY_MODEL = "claude-e2e-relay-only-7" OAUTH_TOKEN = "sk-ant-oat01-e2e-fake-oauth-token" -def known(key: str): - reason, signature = KNOWN[key] - return pytest.mark.xfail(strict=True, raises=signature, reason=reason) - - @pytest.fixture def rig(tmp_path: Path): made: list[Rig] = [] @@ -96,20 +95,18 @@ def test_model_discovery_serves_the_live_native_catalog(rig) -> None: assert RELAY_ONLY_MODEL in ids, f"live catalog not served: {ids[:15]}" -@pytest.mark.parametrize("via", [ - pytest.param("env", marks=known("discovery_env")), - "config", -]) +@pytest.mark.parametrize("via", ["env", "config"]) def test_model_discovery_probes_the_configured_anthropic_endpoint(rig, via: str) -> None: relay = AnthropicMessagesServer([], models=[RELAY_ONLY_MODEL, "claude-e2e-relay-other"]).start() try: r = rig([]) ids = _discover(r, relay, via=via) probed = [g["path"] for g in relay.gets if "/v1/models" in g["path"]] - if not probed or RELAY_ONLY_MODEL not in ids: - raise DiscoveryIgnoredEndpoint( - f"configured endpoint probed={bool(probed)} (native host saw {len(r.srv.gets)} GETs); " - f"relay catalog served={RELAY_ONLY_MODEL in ids}: {ids[:15]}") + with known_gate(KNOWN, f"discovery_{via}", raises=DiscoveryIgnoredEndpoint): + if not probed or RELAY_ONLY_MODEL not in ids: + raise DiscoveryIgnoredEndpoint( + f"configured endpoint probed={bool(probed)} (native host saw {len(r.srv.gets)} GETs); " + f"relay catalog served={RELAY_ONLY_MODEL in ids}: {ids[:15]}") finally: relay.stop() @@ -155,7 +152,6 @@ def test_oauth_wire_names_map_back_to_real_tools(rig) -> None: assert not r.srv.schema_errors(), r.srv.schema_errors() -@known("alias_describe") def test_oauth_deferred_alias_resolves_through_tool_describe(rig) -> None: """The deferred catalog advertises ``chat_history_lookup`` (the OAuth alias of ``session_search``); asking ``tool_describe`` for that exact name must return its schema.""" @@ -169,5 +165,6 @@ def test_oauth_deferred_alias_resolves_through_tool_describe(rig) -> None: (result,) = _tool_results(mains[1]["body"]) not_found = "not_found" in result and "chat_history_lookup" in result.split("not_found", 1)[1][:80] described = "chat_history_lookup" in result and ("parameters" in result or "input_schema" in result) - if not_found or not described: - raise AliasNotReverseMapped(f"tool_describe did not resolve the advertised alias: {result[:600]}") + with known_gate(KNOWN, "alias_describe", raises=AliasNotReverseMapped): + if not_found or not described: + raise AliasNotReverseMapped(f"tool_describe did not resolve the advertised alias: {result[:600]}") diff --git a/tests/e2e/core/providers/test_anthropic_errors.py b/tests/e2e/core/providers/test_anthropic_errors.py index 31ade10169985..952bad2c87a4c 100644 --- a/tests/e2e/core/providers/test_anthropic_errors.py +++ b/tests/e2e/core/providers/test_anthropic_errors.py @@ -15,6 +15,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.providers._anthropic_helpers import Rig, dump, normalised, start_rig, thinking_of from tests.fakes.providers.anthropic_messages import ApiError, DropStream, Reply, Text, Thinking @@ -25,21 +26,17 @@ class StreamDropAcceptedAsAnswer(AssertionError): """#121320's signature, raised ONLY where the user is handed the dropped stream's fragment. - Strict xfails below accept nothing else, so a harness failure (process death, timeout, a + It is the type ``known_gate`` accepts below, so a harness failure (process death, timeout, a later assertion once the bug is fixed) still fails the run.""" -# Red on current main for a tracked, open bug. Strict: the test FAILS the moment the bug is -# fixed, so the entry is removed with the fix instead of masking a later regression. -KNOWN: dict[str, str] = { - "stream_drop_retry": "#121320 stream closed before message_stop is accepted as a complete answer", +# Red on current main for a tracked, open bug: key -> (the bug's own failure-message pattern, reason). +KNOWN: dict[str, tuple[str, str]] = { + "stream_drop_retry": (r"user got .*(PARTIAL-THOUGHT|FRAGMENT).* instead of exactly one '(RECOVERED-ANSWER|TURN-ONE)'", + "#121320 stream closed before message_stop is accepted as a complete answer"), } -def known(key: str) -> pytest.MarkDecorator: - return pytest.mark.xfail(strict=True, raises=StreamDropAcceptedAsAnswer, reason=KNOWN[key]) - - def _answer_must_be(stdout: str, expected: str, fragment: str) -> None: if stdout.count(expected) != 1 or fragment in stdout: raise StreamDropAcceptedAsAnswer(f"user got {stdout[-800:]!r} instead of exactly one {expected!r}") @@ -106,7 +103,6 @@ def test_400_invalid_request_is_surfaced_without_a_retry_storm(rig_factory) -> N assert vendor_message in surfaced, f"the vendor's invalid_request_error message never reached the user: {surfaced[-800:]}" -@known("stream_drop_retry") def test_stream_drop_mid_thinking_retries_without_duplicate_persisted_content(rig_factory) -> None: """The socket dies after two thinking deltas (no signature, no message_stop). The retry must resend the same history (no half-streamed assistant turn leaks into it), the user sees the @@ -118,7 +114,8 @@ def test_stream_drop_mid_thinking_retries_without_duplicate_persisted_content(ri ], config={"agent": {"api_max_retries": 2}}) proc = rig.run("chat", "-q", "hello", "-Q") assert proc.returncode == 0, proc.stderr[-2000:] - _answer_must_be(proc.stdout, "RECOVERED-ANSWER", "PARTIAL-THOUGHT") + with known_gate(KNOWN, "stream_drop_retry", raises=StreamDropAcceptedAsAnswer): + _answer_must_be(proc.stdout, "RECOVERED-ANSWER", "PARTIAL-THOUGHT") assert "LOST-ANSWER" not in proc.stdout, proc.stdout[-800:] mains = [r["body"] for r in rig.srv.main_requests()] assert len(mains) == 2, [r.get("response") for r in rig.srv.requests] @@ -134,7 +131,6 @@ def test_stream_drop_mid_thinking_retries_without_duplicate_persisted_content(ri assert persisted.count("Clean retry reasoning.") >= 1 -@known("stream_drop_retry") def test_stream_drop_then_next_turn_replays_only_the_completed_signature(rig_factory) -> None: """After a mid-thinking drop + successful retry, the NEXT user turn replays the retry's signed thinking byte-exact — never the dropped stream's unsigned fragment.""" @@ -145,7 +141,8 @@ def test_stream_drop_then_next_turn_replays_only_the_completed_signature(rig_fac ], config={"agent": {"api_max_retries": 2}}) first = rig.run("chat", "-q", "one", "-Q") assert first.returncode == 0, first.stderr[-2000:] - _answer_must_be(first.stdout, "TURN-ONE", "FRAGMENT") + with known_gate(KNOWN, "stream_drop_retry", raises=StreamDropAcceptedAsAnswer): + _answer_must_be(first.stdout, "TURN-ONE", "FRAGMENT") (session_id,) = rig.session_ids() second = rig.run("chat", "--resume", session_id, "-q", "two", "-Q") assert second.returncode == 0 and "TURN-TWO" in second.stdout, second.stderr[-2000:] diff --git a/tests/e2e/core/providers/test_chat_custom_endpoint.py b/tests/e2e/core/providers/test_chat_custom_endpoint.py index 132e3cf1c39e3..82cf1d477a2b9 100644 --- a/tests/e2e/core/providers/test_chat_custom_endpoint.py +++ b/tests/e2e/core/providers/test_chat_custom_endpoint.py @@ -31,7 +31,6 @@ db_tool_calls, db_messages, bounded_turn, - known_marks, oneshot, ) from tests.fakes.providers.chat_variants import CDropToolCall, CText, CTools, FakeChatVariantServer @@ -40,15 +39,14 @@ STRICT_TURN_BUDGET = 60.0 -KNOWN: dict[str, str] = { - "unrepairable_args_surfaced": "#119389 unrepairable tool_call arguments silently dropped, turn reports success", +KNOWN: dict[str, tuple[str, str]] = { + "unrepairable_args_surfaced": ( + r"the unparseable write_file call vanished: the next request never told the model it did not run " + r"and the turn reported success", + "#119389 unrepairable tool_call arguments silently dropped, turn reports success"), } -def known(name: str) -> list: - return known_marks(KNOWN, name) - - def test_ollama_strict_endpoint_always_receives_the_user_turn(tmp_path) -> None: h = Home(tmp_path) script = [ @@ -87,8 +85,7 @@ def test_ollama_strict_endpoint_always_receives_the_user_turn(tmp_path) -> None: assert persisted == [READ_TOOL] * 3, persisted -@pytest.mark.parametrize("finish", [pytest.param("stop", marks=known("unrepairable_args_surfaced")), - pytest.param("tool_calls", marks=known("unrepairable_args_surfaced"))]) +@pytest.mark.parametrize("finish", ["stop", "tool_calls"]) def test_unrepairable_tool_arguments_are_surfaced_not_dropped(tmp_path, finish) -> None: h = Home(tmp_path) broken = '{"path": "out.md", "content": "# Title\\n\\nsays "quoted" and then, ' # unterminated, bad quotes @@ -103,7 +100,7 @@ def test_unrepairable_tool_arguments_are_surfaced_not_dropped(tmp_path, finish) m.get("role") == "tool" or "write_file" in str(m.get("content") or "") for m in mains[1]["messages"][len(mains[0]["messages"]):]) failed_visibly = run.proc.returncode != 0 or run.usage.get("failed") - with bug_assertions(): + with bug_assertions(KNOWN, "unrepairable_args_surfaced"): assert told or failed_visibly, ( "the unparseable write_file call vanished: the next request never told the model it did not run " f"and the turn reported success ({run.stdout.strip()!r}); out.md={(h.project / 'out.md').read_text()!r}") diff --git a/tests/e2e/core/providers/test_chat_errors.py b/tests/e2e/core/providers/test_chat_errors.py index c3c434502d1f9..c8c2b827be942 100644 --- a/tests/e2e/core/providers/test_chat_errors.py +++ b/tests/e2e/core/providers/test_chat_errors.py @@ -28,7 +28,6 @@ custom_chat_config, db_messages, db_tool_calls, - known_marks, oneshot, tool_call_args, ) @@ -43,10 +42,12 @@ pytestmark = pytest.mark.skipif(not sys.platform.startswith("linux"), reason="subprocess harness is Linux-gated") -# Scenario -> "#issue one-line symptom" for scenarios red on origin/main (strict xfail that -# only a KnownBugError from bug_assertions() satisfies; see known_marks). -KNOWN: dict[str, str] = { - "in_stream_ban_fails_once": "#121270 in-stream error code 403 ignored: ban retried, reported as temporarily unavailable", +# Scenario -> (pattern, "#issue one-line symptom") for scenarios red on origin/main: a +# KnownBugError from bug_assertions() matching the pattern XFAILs the cell (known_gate). +KNOWN: dict[str, tuple[str, str]] = { + "in_stream_ban_fails_once": ( + r"(?s)\d+ requests for a permanent account ban: .*temporarily unavailable", + "#121270 in-stream error code 403 ignored: ban retried, reported as temporarily unavailable"), } DOTENV = {"OPENAI_API_KEY": "sk-fake"} @@ -56,10 +57,6 @@ EPSILON = 0.1 -def known(name: str) -> list: - return known_marks(KNOWN, name) - - def _home(tmp_path, srv: FakeChatVariantServer, **config) -> Home: cfg = custom_chat_config(srv.base_url) cfg.update(config) @@ -138,8 +135,7 @@ def test_stream_drop_mid_tool_call_executes_tool_once(tmp_path) -> None: "metadata": {"provider_name": "UpstreamCo", "raw": json.dumps({"error": "account banned"})}} -@pytest.mark.parametrize("_", [pytest.param(None, id="ban", marks=known("in_stream_ban_fails_once"))]) -def test_in_stream_upstream_ban_fails_once_and_visibly(tmp_path, _) -> None: +def test_in_stream_upstream_ban_fails_once_and_visibly(tmp_path) -> None: script: list = [CStreamError(BAN) for _ in range(6)] with FakeChatVariantServer(script, default_text="SHOULD-NOT-ANSWER") as srv: h = _home(tmp_path, srv) @@ -147,7 +143,7 @@ def test_in_stream_upstream_ban_fails_once_and_visibly(tmp_path, _) -> None: records = srv.main_records() assert records, f"precondition: the endpoint was reached: {run.describe()}" - with bug_assertions(): + with bug_assertions(KNOWN, "in_stream_ban_fails_once"): assert "SHOULD-NOT-ANSWER" not in run.stdout, run.describe() assert run.proc.returncode != 0 or run.usage.get("failed") is True, run.describe() surfaced = (run.stdout + run.proc.stderr).lower() diff --git a/tests/e2e/core/providers/test_chat_reasoning_variants.py b/tests/e2e/core/providers/test_chat_reasoning_variants.py index 06c6c98a85f5b..90e995bb6cd95 100644 --- a/tests/e2e/core/providers/test_chat_reasoning_variants.py +++ b/tests/e2e/core/providers/test_chat_reasoning_variants.py @@ -28,7 +28,6 @@ chat_messages, custom_chat_config, db_messages, - known_marks, oneshot, ) from tests.e2e.core.providers._openai_tui import TuiGateway @@ -36,16 +35,13 @@ pytestmark = pytest.mark.skipif(not sys.platform.startswith("linux"), reason="subprocess harness is Linux-gated") -KNOWN: dict[str, str] = { - "reasoning_budget_session_keeps_answering": - "#118182 replayed reasoning_details grow past a route budget and wedge the session on a 400", +KNOWN: dict[str, tuple[str, str]] = { + "reasoning_budget_session_keeps_answering": ( + r"turns \[\d[\d, ]*\] were not answered once replayed reasoning passed the budget", + "#118182 replayed reasoning_details grow past a route budget and wedge the session on a 400"), } -def known(name: str) -> list: - return known_marks(KNOWN, name) - - def _rd(tag: str, text: str | None = None) -> list[dict]: return [{"type": "reasoning.text", "text": text or f"thinking {tag}", "signature": f"sig-{tag}", "format": "anthropic-claude-v1", "index": 0}] @@ -128,8 +124,7 @@ def test_reasoning_content_echoed_on_tool_call_messages(tmp_path) -> None: assert _assistant_field(mains[2], "reasoning_content") == ["ds-1", "ds-2"], chat_messages(mains[2], "assistant") -@pytest.mark.parametrize("scenario", [pytest.param("budget", marks=known("reasoning_budget_session_keeps_answering"))]) -def test_long_session_does_not_wedge_on_replayed_reasoning_budget(tmp_path, scenario) -> None: +def test_long_session_does_not_wedge_on_replayed_reasoning_budget(tmp_path) -> None: """Each turn mints ~1 KB of reasoning; the route 400s ("Provider returned error", not retryable) once the replayed total passes 4 KB. Every turn must still be answered.""" turns = 8 @@ -150,5 +145,5 @@ def test_long_session_does_not_wedge_on_replayed_reasoning_budget(tmp_path, scen rejected = [i for i, r in enumerate(records) if r.get("response") == "route_rejection"] assert rejected, "precondition: the replayed total crossed the route budget at least once" missing = [i for i in range(turns) if f"ANSWER-{i}" not in answers[i]] - with bug_assertions(): + with bug_assertions(KNOWN, "reasoning_budget_session_keeps_answering"): assert not missing, f"turns {missing} were not answered once replayed reasoning passed the budget: {answers}" diff --git a/tests/e2e/core/providers/test_fallback_providers.py b/tests/e2e/core/providers/test_fallback_providers.py index dd7527d43f19e..3d3de656c2c8c 100644 --- a/tests/e2e/core/providers/test_fallback_providers.py +++ b/tests/e2e/core/providers/test_fallback_providers.py @@ -36,31 +36,28 @@ chat_messages, custom_chat_config, db_messages, - known_marks, ) from tests.fakes.fake_llm_provider import FakeLLMServer from tests.fakes.providers.chat_variants import CError, CText, FakeChatVariantServer pytestmark = pytest.mark.skipif(not sys.platform.startswith("linux"), reason="subprocess harness is Linux-gated") -# Scenario -> "#issue one-line symptom" for scenarios red on origin/main (strict xfail that -# only a KnownBugError from bug_assertions() satisfies; see known_marks). -KNOWN: dict[str, str] = { - "unreachable_portal_falls_back": "#120608 transport error during credential resolution skips fallback_providers", +# Scenario -> (pattern, "#issue one-line symptom") for scenarios red on origin/main: a +# KnownBugError from bug_assertions() matching the pattern XFAILs the cell (known_gate). +KNOWN: dict[str, tuple[str, str]] = { + "unreachable_portal_falls_back": ( + r"(?s)fallback_providers never consulted: .*agent failed: \[Errno 111\] Connection refused", + "#120608 transport error during credential resolution skips fallback_providers"), } FALLBACK_MODEL = "fallback-model" PROMPT = "CANARY-PROMPT say hello" FIRST_PROMPT = "CANARY-FIRST what is two plus two" # A primary that is never abandoned for the fallback keeps backing off for minutes, so a -# turn overrunning this is a HarnessError (a real failure even under a KNOWN xfail). +# turn overrunning this is a HarnessError (a real failure even under a KNOWN entry). TURN_BUDGET = 45.0 -def known(name: str) -> list: - return known_marks(KNOWN, name) - - def _fallback_entry(fallback: FakeLLMServer) -> list[dict]: return [{"provider": "custom", "model": FALLBACK_MODEL, "base_url": fallback.base_url}] @@ -179,7 +176,7 @@ def _portal(kind: str) -> Iterator[tuple[str, list[str]]]: @pytest.mark.parametrize("portal_kind", [ pytest.param("5xx", id="portal_5xx"), - pytest.param("refused", id="portal_refused", marks=known("unreachable_portal_falls_back")), + pytest.param("refused", id="portal_refused"), ]) def test_primary_credential_resolution_failure_falls_back(tmp_path, portal_kind: str) -> None: dead = f"http://127.0.0.1:{_closed_port()}" @@ -198,7 +195,8 @@ def test_primary_credential_resolution_failure_falls_back(tmp_path, portal_kind: if portal_kind == "5xx": assert "/api/oauth/token" in portal_hits, f"precondition: the expired token was never refreshed: {run.describe()}" - with bug_assertions(): + scenario = "unreachable_portal_falls_back" if portal_kind == "refused" else f"portal_{portal_kind}" + with bug_assertions(KNOWN, scenario): assert fallback_mains, f"fallback_providers never consulted: {run.describe()}" assert any(PROMPT in t for t in _user_texts(fallback_mains[0])), fallback_mains[0].get("messages") assert run.proc.returncode == 0 and run.stdout.strip() == "FROM-FALLBACK", run.describe() diff --git a/tests/e2e/core/providers/test_native_bedrock_converse_faults.py b/tests/e2e/core/providers/test_native_bedrock_converse_faults.py index c822abe011759..0adb9cb2ae907 100644 --- a/tests/e2e/core/providers/test_native_bedrock_converse_faults.py +++ b/tests/e2e/core/providers/test_native_bedrock_converse_faults.py @@ -18,6 +18,7 @@ pytest.importorskip("botocore") +from tests.e2e.core._pending_fixes import known_gate # noqa: E402 from tests.e2e.core.providers._native_helpers import ( # noqa: E402 ChatResult, KnownSymptom, NativeHome, assert_no_duplicate_assistant_text, make_home, messages, run_chat, ) @@ -34,9 +35,12 @@ "bedrock:InvokeModelWithResponseStream on resource: arn:aws:bedrock:us-east-1::foundation-model/" + MODEL) -KNOWN = { - "validation_retried": "#121294 a Bedrock 400 ValidationException is retried and reported as 'temporarily unavailable'", - "eof_before_message_stop": "#109988 a ConverseStream that ends before messageStop is accepted as the answer", +# Red on current main for a tracked, open bug: key -> (the bug's own failure-message pattern, reason). +KNOWN: dict[str, tuple[str, str]] = { + "validation_retried": (r"^400 ValidationException retried: fake saw [2-9]\d* requests", + "#121294 a Bedrock 400 ValidationException is retried and reported as 'temporarily unavailable'"), + "eof_before_message_stop": (r"^truncated stream accepted as the answer: requests=1 rows=", + "#109988 a ConverseStream that ends before messageStop is accepted as the answer"), } @@ -132,7 +136,6 @@ def test_mid_stream_drop_retries_without_duplicated_persisted_content(runs: dict assert run["requests"][0]["reply"] == "Drop", run["requests"][0] -@pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN["eof_before_message_stop"]) def test_stream_ending_before_message_stop_is_not_accepted(runs: dict[str, Any]) -> None: run = runs["eof_before_message_stop"] assert run["result"].returncode == 0, run["result"].describe() @@ -140,20 +143,21 @@ def test_stream_ending_before_message_stop_is_not_accepted(runs: dict[str, Any]) rows = [r["content"] for r in _assistant_rows(run["nh"])] sent = len(run["requests"]) # Symptom: the truncated first stream is persisted as the final answer and never retried. - if sent == 1 and rows and FINAL.startswith(rows[-1]) and rows[-1] != FINAL: - raise KnownSymptom(f"{KNOWN['eof_before_message_stop']}: requests={sent} rows={rows}") + with known_gate(KNOWN, "eof_before_message_stop", raises=KnownSymptom): + if sent == 1 and rows and FINAL.startswith(rows[-1]) and rows[-1] != FINAL: + raise KnownSymptom(f"truncated stream accepted as the answer: requests={sent} rows={rows}") assert (sent, rows) == (2, [FINAL]), f"requests={sent} rows={rows}" -@pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN["validation_retried"]) def test_validation_exception_is_surfaced_once_without_retry(runs: dict[str, Any]) -> None: run = runs["validation"] result: ChatResult = run["result"] sent = len(run["requests"]) assert sent >= 1 and run["requests"][0]["reply"] == "HttpError", run["requests"] # Symptom: the 400 is retried (the scripted success behind it is reached). - if sent > 1: - raise KnownSymptom(f"{KNOWN['validation_retried']}: fake saw {sent} requests") + with known_gate(KNOWN, "validation_retried", raises=KnownSymptom): + if sent > 1: + raise KnownSymptom(f"400 ValidationException retried: fake saw {sent} requests") shown = result.stdout + result.stderr assert result.returncode != 0 and FINAL not in shown, result.describe() assert VALIDATION_MARK in shown and "ValidationException" in shown, result.describe() diff --git a/tests/e2e/core/providers/test_native_bedrock_converse_turns.py b/tests/e2e/core/providers/test_native_bedrock_converse_turns.py index c95de23f6fd76..70e87f1a082ff 100644 --- a/tests/e2e/core/providers/test_native_bedrock_converse_turns.py +++ b/tests/e2e/core/providers/test_native_bedrock_converse_turns.py @@ -20,6 +20,7 @@ pytest.importorskip("botocore") +from tests.e2e.core._pending_fixes import known_gate # noqa: E402 from tests.e2e.core.providers._native_helpers import ( # noqa: E402 ChatResult, KnownSymptom, NativeHome, latest_session, make_home, messages, run_chat, session_ids, tool_calls_of, ) @@ -29,9 +30,12 @@ MODEL = "deepseek.v3-v1:0" -KNOWN = { - "reasoning_shredded": "#98468 streamed reasoning is persisted with '\\n\\n' between every delta", - "resume_drops_reasoning": "#121293 --resume replays Bedrock assistant turns without their signed reasoningContent", +# Red on current main for a tracked, open bug: key -> (the bug's own failure-message pattern, reason). +KNOWN: dict[str, tuple[str, str]] = { + "reasoning_shredded": (r"^persisted reasoning has blank lines between streamed deltas: ", + "#98468 streamed reasoning is persisted with '\\n\\n' between every delta"), + "resume_drops_reasoning": (r"^resumed assistant tool-use turn replayed without signed reasoningContent: ", + "#121293 --resume replays Bedrock assistant turns without their signed reasoningContent"), } SEED_TEXT = "codeword PELICAN-5501" @@ -225,13 +229,13 @@ def _reasoning_rows(nh: NativeHome) -> list[str]: return [r["reasoning_content"] for r in messages(nh) if r["role"] == "assistant" and r["reasoning_content"]] -@pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN["reasoning_shredded"]) def test_persisted_reasoning_equals_the_streamed_reasoning_text(runs: dict[str, Any]) -> None: _ok(runs["tools"]["result"]) persisted = _reasoning_rows(runs["tools"]["nh"]) assert [p.replace("\n\n", "") for p in persisted] == [R_A1, R_A2, R_A3], persisted - if persisted != [R_A1, R_A2, R_A3]: # same text once the blank lines are removed: exactly the bug - raise KnownSymptom(f"{KNOWN['reasoning_shredded']}: {persisted}") + with known_gate(KNOWN, "reasoning_shredded", raises=KnownSymptom): + if persisted != [R_A1, R_A2, R_A3]: # same text once the blank lines are removed: exactly the bug + raise KnownSymptom(f"persisted reasoning has blank lines between streamed deltas: {persisted}") # -------------------------------------------------------------------------------------------------- @@ -258,7 +262,6 @@ def test_resume_in_new_process_replays_tool_history_valid_for_converse(runs: dic assert rows[-1]["content"] == FINAL_B2 -@pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN["resume_drops_reasoning"]) def test_resume_replays_signed_reasoning_verbatim(runs: dict[str, Any]) -> None: run = runs["resume"] _ok(run["first"]) @@ -271,8 +274,10 @@ def test_resume_replays_signed_reasoning_verbatim(runs: dict[str, Any]) -> None: assert requests[1]["body"]["messages"][1]["content"][0] == signed[0] # ... and after --resume (new process) it must be identical. resumed = requests[2]["body"]["messages"] - if not [b for b in resumed[1]["content"] if "reasoningContent" in b]: - raise KnownSymptom(f"{KNOWN['resume_drops_reasoning']}: {resumed[1]['content']}") + with known_gate(KNOWN, "resume_drops_reasoning", raises=KnownSymptom): + if not [b for b in resumed[1]["content"] if "reasoningContent" in b]: + raise KnownSymptom(f"resumed assistant tool-use turn replayed without signed reasoningContent: " + f"{resumed[1]['content']}") assert resumed[1]["content"][0] == signed[0], resumed[1]["content"] assert [b for b in resumed[3]["content"] if "reasoningContent" in b] == [ b for b in requests[1]["emitted"] if "reasoningContent" in b] diff --git a/tests/e2e/core/providers/test_native_codex_app_server.py b/tests/e2e/core/providers/test_native_codex_app_server.py index 6e6fdfd3f9b7f..fb1ec492e9fcb 100644 --- a/tests/e2e/core/providers/test_native_codex_app_server.py +++ b/tests/e2e/core/providers/test_native_codex_app_server.py @@ -19,6 +19,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.providers._native_helpers import KnownSymptom, messages, tool_calls_of from tests.fakes.providers.codex_app_server import CodexRun, run_codex_scenario @@ -28,8 +29,10 @@ pytest.mark.live_system_guard_bypass, ] -KNOWN = { - "compaction_row": "#121301 native contextCompaction item persisted as a raw-JSON assistant message", +# Red on current main for a tracked, open bug: key -> (the bug's own failure-message pattern, reason). +KNOWN: dict[str, tuple[str, str]] = { + "compaction_row": (r"^compaction boundary persisted as assistant content: \[.*contextCompaction", + "#121301 native contextCompaction item persisted as a raw-JSON assistant message"), } YOLO = ["--yolo"] @@ -195,12 +198,12 @@ def test_native_compaction_keeps_thread_and_transcript(runs): assert texts == ["C-ONE", "C-TWO", "C-THREE"], f"transcript rewritten or duplicated: {rows}" -@pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN["compaction_row"]) def test_native_compaction_is_not_persisted_as_assistant_text(runs): run = runs["compact"] assert [r.returncode for r in run.results] == [0, 0], "\n".join(r.describe() for r in run.results) rows = _rows(run) assert any(r["content"] == "C-TWO" for r in rows), f"post-compaction answer not persisted: {rows}" leaked = [r["content"] for r in rows if "contextCompaction" in (r.get("content") or "")] - if leaked: - raise KnownSymptom(f"compaction boundary persisted as assistant content: {leaked}") + with known_gate(KNOWN, "compaction_row", raises=KnownSymptom): + if leaked: + raise KnownSymptom(f"compaction boundary persisted as assistant content: {leaked}") diff --git a/tests/e2e/core/providers/test_native_codex_app_server_faults.py b/tests/e2e/core/providers/test_native_codex_app_server_faults.py index 0629e06f25b1a..1b5c098fd6a74 100644 --- a/tests/e2e/core/providers/test_native_codex_app_server_faults.py +++ b/tests/e2e/core/providers/test_native_codex_app_server_faults.py @@ -13,6 +13,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.providers._native_helpers import KnownSymptom, messages, wait_until from tests.fakes.providers.codex_app_server import CodexRun, pid_alive, run_codex_scenario @@ -23,11 +24,17 @@ pytest.mark.live_system_guard_bypass, ] -KNOWN = { - "q_approval": "#121296 approval in `chat -q` waits the full approvals.timeout instead of single_query_mode", - "permissions": "#121297 reply to item/permissions/requestApproval omits required `permissions`", - "orphan": "#121298 `chat -q` exit never closes the codex session; own-session descendants orphaned", - "failed_hidden": "#121299 failed turn after an agentMessage prints the message and hides the reason", +# Red on current main for a tracked, open bug: key -> (the bug's own failure-message pattern, reason). +KNOWN: dict[str, tuple[str, str]] = { + "q_approval": (r"^approval parked \d+\.\ds on a prompt nobody can answer in -q", + "#121296 approval in `chat -q` waits the full approvals.timeout instead of single_query_mode"), + "permissions": (r"^PermissionsRequestApprovalResponse without `permissions`: .*'violation': 'missing field `permissions`'", + "#121297 reply to item/permissions/requestApproval omits required `permissions`"), + # The poll itself raises the symptom; anchor on its own subject so no other wait can match. + "orphan": (r"^timed out after [\d.]+s waiting for app-server descendant \d+ to be reaped after CLI exit", + "#121298 `chat -q` exit never closes the codex session; own-session descendants orphaned"), + "failed_hidden": (r"^turn failure reason never shown to the user: ", + "#121299 failed turn after an agentMessage prints the message and hides the reason"), } YOLO = ["--yolo"] @@ -103,15 +110,14 @@ def test_will_retry_error_notification_is_not_terminal(runs): assert _assistant_texts(run) == ["RETRY-OK"] -@pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN["failed_hidden"]) def test_failed_turn_after_agent_message_surfaces_reason(runs): run = runs["failed_hidden"] assert run.results[0].returncode != 0 and "PARTIAL-B" in run.results[0].stdout, run.results[0].describe() - if "FAIL-MARKER-89" not in run.output: - raise KnownSymptom(f"turn failure reason never shown to the user: {run.output!r}") + with known_gate(KNOWN, "failed_hidden", raises=KnownSymptom): + if "FAIL-MARKER-89" not in run.output: + raise KnownSymptom(f"turn failure reason never shown to the user: {run.output!r}") -@pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN["q_approval"]) def test_single_query_approval_resolves_without_waiting_for_a_human(runs): run = runs["q_approval"] entries = run.fake.entries() @@ -122,26 +128,27 @@ def test_single_query_approval_resolves_without_waiting_for_a_human(runs): reply = replies[0] assert reply["msg"]["id"] == sent[0]["msg"]["id"] and not reply.get("violation"), reply waited = reply["t"] - sent[0]["t"] - if waited >= APPROVAL_TIMEOUT_S / 2: - raise KnownSymptom(f"approval parked {waited:.1f}s on a prompt nobody can answer in -q") + with known_gate(KNOWN, "q_approval", raises=KnownSymptom): + if waited >= APPROVAL_TIMEOUT_S / 2: + raise KnownSymptom(f"approval parked {waited:.1f}s on a prompt nobody can answer in -q") -@pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN["permissions"]) def test_permissions_request_reply_matches_protocol(runs): run = runs["permissions"] replies = run.fake.replies_to("item/permissions/requestApproval") assert len(replies) == 1, f"permissions request unanswered: {replies}" violation = replies[0].get("violation") or "" - if "permissions" in violation: - raise KnownSymptom(f"PermissionsRequestApprovalResponse without `permissions`: {replies[0]}") + with known_gate(KNOWN, "permissions", raises=KnownSymptom): + if "permissions" in violation: + raise KnownSymptom(f"PermissionsRequestApprovalResponse without `permissions`: {replies[0]}") assert not violation, f"invalid PermissionsRequestApprovalResponse: {replies[0]}" -@pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN["orphan"]) def test_cli_exit_reaps_app_server_descendants(runs): run = runs["orphan"] assert run.results[0].returncode == 0 and "REAP-DONE" in run.results[0].stdout, run.results[0].describe() children = run.fake.grandchild_pids() assert len(children) == 1, f"the fake must have spawned exactly one descendant: {children}" - wait_until(lambda: not pid_alive(children[0]), 3.0, - f"app-server descendant {children[0]} to be reaped after CLI exit", error=KnownSymptom) + with known_gate(KNOWN, "orphan", raises=KnownSymptom): + wait_until(lambda: not pid_alive(children[0]), 3.0, + f"app-server descendant {children[0]} to be reaped after CLI exit", error=KnownSymptom) diff --git a/tests/e2e/core/providers/test_native_copilot_acp.py b/tests/e2e/core/providers/test_native_copilot_acp.py index ac0bdbbf62eaf..df9d8864a4913 100644 --- a/tests/e2e/core/providers/test_native_copilot_acp.py +++ b/tests/e2e/core/providers/test_native_copilot_acp.py @@ -34,6 +34,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.providers._native_helpers import ( ChatResult, KnownSymptom, @@ -74,8 +75,10 @@ FINAL_COMPACT = "All eight files read (FINAL-COMPACT)." -KNOWN: dict[str, str] = { - "late_chunk": "#65788 agent_message_chunk emitted after the session/prompt result is dropped", +# Red on current main for a tracked, open bug: key -> (the bug's own failure-message pattern, reason). +KNOWN: dict[str, tuple[str, str]] = { + "late_chunk": (r"^late chunk dropped: \d+ model calls, stdout=", + "#65788 agent_message_chunk emitted after the session/prompt result is dropped"), } @@ -300,14 +303,14 @@ def test_compaction_in_an_acp_session_keeps_the_next_prompt_valid_and_grounded(o assert any(FINAL_COMPACT in (r["content"] or "") for r in rows if r["role"] == "assistant"), rows -@pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN["late_chunk"]) def test_message_chunk_after_prompt_result_reaches_the_user(outcomes): sc = outcomes["late"] run = sc.runs[0] assert run.returncode == 0, run.describe() assert sc.fake.invalid() == [], sc.fake.invalid() rows = messages(sc.nh, latest_session(sc.nh)) - if LATE_TEXT not in run.stdout: - raise KnownSymptom(f"late chunk dropped: {len(sc.fake.main_prompts())} model calls, stdout={run.stdout!r}") + with known_gate(KNOWN, "late_chunk", raises=KnownSymptom): + if LATE_TEXT not in run.stdout: + raise KnownSymptom(f"late chunk dropped: {len(sc.fake.main_prompts())} model calls, stdout={run.stdout!r}") assert_no_duplicate_assistant_text(rows, LATE_TEXT) assert len(sc.fake.main_prompts()) == 1, "a delivered late chunk must not trigger an empty-response retry" diff --git a/tests/e2e/core/providers/test_native_copilot_acp_errors.py b/tests/e2e/core/providers/test_native_copilot_acp_errors.py index 62a6515050002..88ecedab99288 100644 --- a/tests/e2e/core/providers/test_native_copilot_acp_errors.py +++ b/tests/e2e/core/providers/test_native_copilot_acp_errors.py @@ -27,7 +27,7 @@ import pytest -from tests.e2e.core._pending_fixes import known_failure +from tests.e2e.core._pending_fixes import known_failure, known_gate from tests.e2e.core.providers._native_helpers import ( ChatResult, KnownSymptom, @@ -53,8 +53,10 @@ CRASH_STDERR = "fatal: agent segfaulted (fake)" -KNOWN: dict[str, str] = { - "auth_remedy": "#121290 copilot-acp auth failure tells the user to run a hermes command that is not implemented", +# Red on current main for a tracked, open bug: key -> (the bug's own failure-message pattern, reason). +KNOWN: dict[str, tuple[str, str]] = { + "auth_remedy": (r"^remedy 'hermes [^']+' is not implemented for copilot-acp: ", + "#121290 copilot-acp auth failure tells the user to run a hermes command that is not implemented"), } @@ -147,7 +149,6 @@ def test_acp_failure_is_retried_per_semantics_and_surfaced_once(outcomes, name): REMEDY_RE = re.compile(r"`(hermes [^`]+)`") -@pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN["auth_remedy"]) def test_auth_failure_remedy_is_an_actionable_command(outcomes): """The sign-in remedy printed for an ACP ``Authentication required`` must not be a dead end: any ``hermes ...`` command it names has to be implemented for this provider.""" @@ -158,5 +159,6 @@ def test_auth_failure_remedy_is_an_actionable_command(outcomes): proc = subprocess.run(argv, cwd=out.nh.project, env=out.nh.env(), capture_output=True, text=True, timeout=60, stdin=subprocess.DEVNULL) said = (proc.stdout + proc.stderr).lower() - if "not implemented" in said: - raise KnownSymptom(f"remedy {command!r} is not implemented for copilot-acp: {said.strip()[:300]}") + with known_gate(KNOWN, "auth_remedy", raises=KnownSymptom): + if "not implemented" in said: + raise KnownSymptom(f"remedy {command!r} is not implemented for copilot-acp: {said.strip()[:300]}") diff --git a/tests/e2e/core/providers/test_native_copilot_acp_streaming.py b/tests/e2e/core/providers/test_native_copilot_acp_streaming.py index 0d5894c8f9c70..a0dbd57ff1250 100644 --- a/tests/e2e/core/providers/test_native_copilot_acp_streaming.py +++ b/tests/e2e/core/providers/test_native_copilot_acp_streaming.py @@ -13,7 +13,7 @@ provider: the outer client must get ``session/update`` chunks before the inner turn ends (#101507). Both are red on main (the ACP client buffers the whole response, then replays it as a stream), so -both are strict xfails that raise :class:`KnownSymptom` only for "no progress before the result". +both are ``known_gate`` cells that XFAIL only on :class:`KnownSymptom` for "no progress before the result". """ from __future__ import annotations @@ -33,6 +33,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.providers._native_helpers import TURN_TIMEOUT, KnownSymptom, NativeHome, make_home from tests.fakes.providers import copilot_acp as acp @@ -49,9 +50,12 @@ MARKERS = (HEAD.strip(), THOUGHT, ANSWER) -KNOWN: dict[str, str] = { - "tui_stream": "#120550 copilot-acp buffers the whole turn: no reasoning/message delta reaches the UI in flight", - "nested_acp": "#101507 hermes acp over copilot-acp forwards inner ACP chunks only after the inner turn ends", +# Red on current main for a tracked, open bug: key -> (the bug's own failure-message pattern, reason). +KNOWN: dict[str, tuple[str, str]] = { + "tui_stream": (r"^tui_stream: no agent chunk reached the surface during the [\d.]+s turn", + "#120550 copilot-acp buffers the whole turn: no reasoning/message delta reaches the UI in flight"), + "nested_acp": (r"^nested_acp: no agent chunk reached the surface during the [\d.]+s turn", + "#101507 hermes acp over copilot-acp forwards inner ACP chunks only after the inner turn ends"), } @@ -211,10 +215,7 @@ def _agent_timeline(fake: acp.AcpFake) -> tuple[float, float]: return first, done -@pytest.mark.parametrize("surface", [ - pytest.param(name, marks=pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN[name])) - for name in SURFACES -]) +@pytest.mark.parametrize("surface", SURFACES) def test_long_reasoning_turn_streams_progress_before_it_completes(observed, surface): obs = observed[surface] assert obs.fake.invalid() == [], f"requests rejected by the ACP schema: {obs.fake.invalid()}" @@ -222,10 +223,11 @@ def test_long_reasoning_turn_streams_progress_before_it_completes(observed, surf first, done = _agent_timeline(obs.fake) assert done - first >= MIN_SPAN_S, f"vacuity: the agent streamed for only {done - first:.2f}s" seen_at = _first_marker_arrival(obs.received) - if seen_at is None or seen_at >= done - 0.3: - late = [round(t - done, 2) for t, m in obs.received if _chunk_text(m)] - raise KnownSymptom(f"{surface}: no agent chunk reached the surface during the {done - first:.1f}s turn; " - f"chunk arrivals relative to the result: {late}") + with known_gate(KNOWN, surface, raises=KnownSymptom): + if seen_at is None or seen_at >= done - 0.3: + late = [round(t - done, 2) for t, m in obs.received if _chunk_text(m)] + raise KnownSymptom(f"{surface}: no agent chunk reached the surface during the {done - first:.1f}s turn; " + f"chunk arrivals relative to the result: {late}") assert seen_at >= first, "a chunk cannot arrive before the agent sent it" diff --git a/tests/e2e/core/providers/test_native_gemini_errors.py b/tests/e2e/core/providers/test_native_gemini_errors.py index 94807db1f814b..f0f5bdb815c82 100644 --- a/tests/e2e/core/providers/test_native_gemini_errors.py +++ b/tests/e2e/core/providers/test_native_gemini_errors.py @@ -19,6 +19,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.providers import _native_helpers as nh from tests.fakes.providers.gemini_native import ( HERMES_ENV, @@ -33,9 +34,10 @@ hermes_model, ) -KNOWN = { - "prompt_blocked": "#121317 promptFeedback.blockReason is retried 9x as an empty stream and reported as " - "'temporarily unavailable'", +KNOWN: dict[str, tuple[str, str]] = { + "prompt_blocked": (r"terminal Google response was re-sent \d+x", + "#121317 promptFeedback.blockReason is retried 9x as an empty stream and reported as " + "'temporarily unavailable'"), } NEVER = "GEMINI-MUST-NOT-BE-SHOWN" @@ -126,19 +128,14 @@ def test_retryable_error_is_retried_then_succeeds(outcomes: dict[str, Outcome], } -def _terminal_param(name: str): - marks = [pytest.mark.xfail(strict=True, raises=nh.KnownSymptom, reason=KNOWN[name])] if name in KNOWN else [] - return pytest.param(name, marks=marks, id=name) - - -@pytest.mark.parametrize("name", [_terminal_param(n) for n in TERMINAL]) +@pytest.mark.parametrize("name", list(TERMINAL)) def test_non_retryable_surfaced_once(outcomes: dict[str, Outcome], name: str) -> None: o = outcomes[name] assert o.calls, f"no request reached the fake\n{o.describe()}" - if len(o.calls) != 1: - # The tracked bug's symptom for KNOWN cases; a plain failure for every other terminal case. - failure = nh.KnownSymptom if name in KNOWN else AssertionError - raise failure(f"terminal Google response was re-sent {len(o.calls)}x\n{o.describe()}") + # The tracked bug's symptom for KNOWN cases; a plain failure for every other terminal case. + with known_gate(KNOWN, name, raises=nh.KnownSymptom): + if len(o.calls) != 1: + raise nh.KnownSymptom(f"terminal Google response was re-sent {len(o.calls)}x\n{o.describe()}") assert o.result.returncode != 0, o.describe() word = TERMINAL[name] lines = [ln.strip() for ln in o.result.stdout.splitlines() if ln.strip()] diff --git a/tests/e2e/core/providers/test_native_gemini_schema.py b/tests/e2e/core/providers/test_native_gemini_schema.py index 1fec461b2b04f..751005acba90c 100644 --- a/tests/e2e/core/providers/test_native_gemini_schema.py +++ b/tests/e2e/core/providers/test_native_gemini_schema.py @@ -23,12 +23,15 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.providers import _native_helpers as nh from tests.fakes.providers.gemini_native import HERMES_ENV, Call, Calls, GeminiFake, Recorded, Text, hermes_model -KNOWN = { - "ref_dropped_v1": "#99438 legacy `parameters` path drops $ref/$defs instead of inlining (empty schema)", - "array_items_v1": "#71804 array parameter without `items` is sent as-is; Google 400s 'items: missing field'", +KNOWN: dict[str, tuple[str, str]] = { + "ref_dropped_v1": (r"\$ref-typed parameter lost its shape on the v1 wire", + "#99438 legacy `parameters` path drops $ref/$defs instead of inlining (empty schema)"), + "array_items_v1": (r"Google rejected the item-less array: .*items: missing field", + "#71804 array parameter without `items` is sent as-is; Google 400s 'items: missing field'"), } TOOL = "mcp__hostile__lookup" @@ -169,19 +172,19 @@ def test_v1_proto_schema_declaration_is_accepted(outcomes: dict[str, Outcome]) - assert params["required"] == ["query"], params -@pytest.mark.xfail(strict=True, raises=nh.KnownSymptom, reason=KNOWN["ref_dropped_v1"]) def test_v1_ref_parameter_keeps_its_shape(outcomes: dict[str, Outcome]) -> None: _assert_round_trip(outcomes["v1"], "v1") mode = outcomes["v1"].declaration()["parameters"]["properties"]["mode"] - if mode == {}: - raise nh.KnownSymptom(f"$ref-typed parameter lost its shape on the v1 wire: {mode}") + with known_gate(KNOWN, "ref_dropped_v1", raises=nh.KnownSymptom): + if mode == {}: + raise nh.KnownSymptom(f"$ref-typed parameter lost its shape on the v1 wire: {mode}") assert mode.get("enum") == ["fast", "slow"], mode -@pytest.mark.xfail(strict=True, raises=nh.KnownSymptom, reason=KNOWN["array_items_v1"]) def test_v1_array_without_items_is_accepted(outcomes: dict[str, Outcome]) -> None: o = outcomes["v1_itemless"] assert o.calls, o.result.describe() - if any("items: missing field" in r for r in o.rejections): - raise nh.KnownSymptom(f"Google rejected the item-less array: {o.rejections}") + with known_gate(KNOWN, "array_items_v1", raises=nh.KnownSymptom): + if any("items: missing field" in r for r in o.rejections): + raise nh.KnownSymptom(f"Google rejected the item-less array: {o.rejections}") _assert_round_trip(o, "v1") diff --git a/tests/e2e/core/providers/test_native_vertex_errors.py b/tests/e2e/core/providers/test_native_vertex_errors.py index 5f8087458fe19..60b7df259a2b0 100644 --- a/tests/e2e/core/providers/test_native_vertex_errors.py +++ b/tests/e2e/core/providers/test_native_vertex_errors.py @@ -21,6 +21,7 @@ pytest.importorskip("google.auth", reason="Vertex minting needs google-auth (CI installs it)") +from tests.e2e.core._pending_fixes import known_gate # noqa: E402 from tests.e2e.core.providers._native_helpers import ChatResult, KnownSymptom, make_home, run_chat # noqa: E402 from tests.fakes.providers.vertex import ( # noqa: E402 PROJECT, @@ -33,11 +34,12 @@ hermes_setup, ) -KNOWN: dict[str, str] = { - "guidance:permission_denied": "#121295 Vertex 401/403 are reported as 'rejected your API key' (Vertex has no API " - "key; the fix is the service account / IAM role)", - "guidance:unauthenticated": "#121295 Vertex 401/403 are reported as 'rejected your API key' (Vertex has no API " - "key; the fix is the service account / IAM role)", +_API_KEY_BLAME = (r"Vertex auth failure blamed on an API key", + "#121295 Vertex 401/403 are reported as 'rejected your API key' (Vertex has no API " + "key; the fix is the service account / IAM role)") +KNOWN: dict[str, tuple[str, str]] = { + "guidance:permission_denied": _API_KEY_BLAME, + "guidance:unauthenticated": _API_KEY_BLAME, } QUOTA_MSG = ("Resource exhausted. Please try again later. Please refer to " @@ -136,11 +138,7 @@ def test_oauth_invalid_grant_sends_nothing_and_names_the_credential(results: dic assert "VERTEX_CREDENTIALS_PATH" in out or "GOOGLE_APPLICATION_CREDENTIALS" in out, turn.describe() -def _guidance_param(name: str) -> Any: - return pytest.param(name, marks=pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN[f"guidance:{name}"])) - - -@pytest.mark.parametrize("name", [_guidance_param("permission_denied"), _guidance_param("unauthenticated")]) +@pytest.mark.parametrize("name", ["permission_denied", "unauthenticated"]) def test_auth_failure_guidance_is_vertex_specific(results: dict[str, Any], name: str) -> None: """Vertex authenticates with an OAuth service account, so the guidance must not send the user to rotate an 'API key' that does not exist.""" @@ -149,5 +147,6 @@ def test_auth_failure_guidance_is_vertex_specific(results: dict[str, Any], name: if not (res["fake"].requests and turn.returncode != 0): raise RuntimeError(f"{name}: the auth failure never happened:\n{turn.describe()}") guidance = _output(turn).split("Provider said:")[0].lower() - if "api key" in guidance: - raise KnownSymptom(f"Vertex auth failure blamed on an API key:\n{turn.stdout}") + with known_gate(KNOWN, f"guidance:{name}", raises=KnownSymptom): + if "api key" in guidance: + raise KnownSymptom(f"Vertex auth failure blamed on an API key:\n{turn.stdout}") diff --git a/tests/e2e/core/providers/test_native_vertex_recovery.py b/tests/e2e/core/providers/test_native_vertex_recovery.py index 7eff31dfc9584..52bb979004309 100644 --- a/tests/e2e/core/providers/test_native_vertex_recovery.py +++ b/tests/e2e/core/providers/test_native_vertex_recovery.py @@ -42,8 +42,6 @@ hermes_setup, ) -KNOWN: dict[str, str] = {} - SUMMARY_MARK = "VERTEX-SUMMARY-OK" TURN1_FINAL = "Compaction turn one done." TURN2_FINAL = "Compaction turn two done." diff --git a/tests/e2e/core/providers/test_native_vertex_tools.py b/tests/e2e/core/providers/test_native_vertex_tools.py index d63048d83d838..9f6481acfc9d8 100644 --- a/tests/e2e/core/providers/test_native_vertex_tools.py +++ b/tests/e2e/core/providers/test_native_vertex_tools.py @@ -25,6 +25,7 @@ pytest.importorskip("google.auth", reason="Vertex minting needs google-auth (CI installs it)") +from tests.e2e.core._pending_fixes import known_gate # noqa: E402 from tests.e2e.core.providers._native_helpers import ( # noqa: E402 ChatResult, KnownSymptom, @@ -46,10 +47,11 @@ signatures_on_wire, ) -# key -> "#issue reason"; a strict xfail turns red the moment the bug is fixed. -KNOWN: dict[str, str] = { - "default_toolset": "#109115 terminal.notify anyOf[boolean, array] is rejected by Vertex's FunctionDeclaration " - "translator, so every default-toolset turn 400s", +# key -> (symptom pattern, "#issue reason"), gated at run time by ``known_gate``. +KNOWN: dict[str, tuple[str, str]] = { + "default_toolset": (r"Vertex rejected the tool declarations: .*schema type should be ARRAY", + "#109115 terminal.notify anyOf[boolean, array] is rejected by Vertex's FunctionDeclaration " + "translator, so every default-toolset turn 400s"), } SECRET_1 = "PINEAPPLE-42" @@ -61,7 +63,7 @@ class Precondition(RuntimeError): - """A scenario broke before reaching the property under test (never masquerades as a KNOWN xfail).""" + """A scenario broke before reaching the property under test (never masquerades as a KNOWN gap).""" def require(ok: Any, what: str) -> None: @@ -235,13 +237,13 @@ def test_expired_token_is_reminted_and_request_retried(results: dict[str, Any]) assert [r["role"] for r in rows] == ["user", "assistant", "tool", "assistant"], rows -@pytest.mark.xfail(strict=True, raises=KnownSymptom, reason=KNOWN["default_toolset"]) def test_default_toolset_schemas_accepted_by_vertex(results: dict[str, Any]) -> None: """With Hermes' default toolsets every tool declaration must survive Vertex's translation.""" res = results["default_toolset"] fake = res["fake"] require(fake.requests and fake.requests[0]["auth"].startswith("Bearer ya29."), "turn never reached Vertex") schema_rejects = [r["rejected"] for r in fake.rejected() if "schema type should be ARRAY" in (r["rejected"] or "")] - if schema_rejects: - raise KnownSymptom(f"Vertex rejected the tool declarations: {schema_rejects[0]}") + with known_gate(KNOWN, "default_toolset", raises=KnownSymptom): + if schema_rejects: + raise KnownSymptom(f"Vertex rejected the tool declarations: {schema_rejects[0]}") assert "Default toolset answer." in res["turn"].stdout, res["turn"].describe() diff --git a/tests/e2e/core/providers/test_oauth_device_flow.py b/tests/e2e/core/providers/test_oauth_device_flow.py index c0446122e8083..9cc9687126155 100644 --- a/tests/e2e/core/providers/test_oauth_device_flow.py +++ b/tests/e2e/core/providers/test_oauth_device_flow.py @@ -13,6 +13,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.providers._oauth_helpers import kill_tagged, make_home, run_hermes from tests.fakes.providers.oauth_token_server import OAuthTokenServer, unsigned_jwt @@ -40,15 +41,20 @@ class Case: class PolledFasterThanAllowed(AssertionError): """Raised ONLY at the poll-cadence assertion, the signature of #121163 / #121254. - The strict xfails accept nothing else: a failed login, a wrong poll count or a timeout is a + It is the type ``known_gate`` accepts: a failed login, a wrong poll count or a timeout is a real failure even in a KNOWN cell.""" -# Red on main for a tracked, open bug. Strict: XPASS fails, forcing the entry out with the fix. -KNOWN: dict[str, str] = { - "honors_server_interval": "#121163 (dup #87432) device-code poll interval is capped to 1s " - "(DEVICE_AUTH_POLL_INTERVAL_CAP_SECONDS used as a ceiling)", - "slow_down_adds_five_seconds": "#121254 slow_down grows the poll interval by 1s, RFC 8628 3.5 requires +5s", +# Red on main for a tracked, open bug: case -> (the bug's own failure-message pattern, reason). +KNOWN: dict[str, tuple[str, str]] = { + "honors_server_interval": ( + r"client polled faster than the server allows \(server interval 3s, .*required at least \[3\.0, 3\.0\]", + "#121163 (dup #87432) device-code poll interval is capped to 1s " + "(DEVICE_AUTH_POLL_INTERVAL_CAP_SECONDS used as a ceiling)"), + "slow_down_adds_five_seconds": ( + r"client polled faster than the server allows \(server interval 1s, .*'slow_down'.*" + r"required at least \[1\.0, 6\.0, 6\.0\]", + "#121254 slow_down grows the poll interval by 1s, RFC 8628 3.5 requires +5s"), } @@ -62,14 +68,7 @@ def expected_min_gaps(case: Case) -> list[float]: return gaps -def _params(): - for name in CASES: - marks = [pytest.mark.xfail(strict=True, raises=PolledFasterThanAllowed, reason=KNOWN[name]) - ] if name in KNOWN else [] - yield pytest.param(name, marks=marks, id=name) - - -@pytest.mark.parametrize("name", list(_params())) +@pytest.mark.parametrize("name", list(CASES)) def test_device_code_login_poll_cadence(name: str, tmp_path) -> None: case = CASES[name] fh = make_home(tmp_path) @@ -98,7 +97,8 @@ def test_device_code_login_poll_cadence(name: str, tmp_path) -> None: gaps = [round(b - a, 3) for a, b in zip(flow.polls, flow.polls[1:])] want = expected_min_gaps(case) short = [(i, got, need) for i, (got, need) in enumerate(zip(gaps, want)) if got + SLACK_S < need] - if short: - raise PolledFasterThanAllowed( - f"client polled faster than the server allows (server interval {case.interval}s, script {case.script}): " - f"observed gaps {gaps}, required at least {want}; short polls (index, got, need): {short}") + with known_gate(KNOWN, name, raises=PolledFasterThanAllowed): + if short: + raise PolledFasterThanAllowed( + f"client polled faster than the server allows (server interval {case.interval}s, script {case.script}): " + f"observed gaps {gaps}, required at least {want}; short polls (index, got, need): {short}") diff --git a/tests/e2e/core/providers/test_openai_codex_pool.py b/tests/e2e/core/providers/test_openai_codex_pool.py index 1bd64eee9eaff..ee5173b4c1d58 100644 --- a/tests/e2e/core/providers/test_openai_codex_pool.py +++ b/tests/e2e/core/providers/test_openai_codex_pool.py @@ -37,7 +37,6 @@ REPO_ROOT, Home, bug_assertions, - known_marks, oneshot, write_sitecustomize_shim, ) @@ -45,15 +44,13 @@ pytestmark = pytest.mark.skipif(not sys.platform.startswith("linux"), reason="subprocess harness is Linux-gated") -KNOWN: dict[str, str] = { - "independent_login_survives": "#120741 terminal refresh failure on one pooled login quarantines an independent login", +KNOWN: dict[str, tuple[str, str]] = { + "independent_login_survives": ( + r"login A was quarantined by B's invalid_grant: live pool is ", + "#120741 terminal refresh failure on one pooled login quarantines an independent login"), } -def known(name: str) -> list: - return known_marks(KNOWN, name) - - TOKEN_URL = "https://auth.openai.com/oauth/token" MODEL = "gpt-5.3-codex" @@ -216,9 +213,7 @@ def _live_order(rows: list[dict[str, Any]]) -> list[tuple[str, str]]: return [(r["id"], r["source"]) for r in sorted(live, key=lambda r: r.get("priority", 0))] -@pytest.mark.parametrize("scenario", [pytest.param("independent_login_survives", - marks=known("independent_login_survives"))]) -def test_dead_grant_on_independent_login_keeps_login_a(tmp_path, scenario) -> None: +def test_dead_grant_on_independent_login_keeps_login_a(tmp_path) -> None: """B's refresh token is terminally rejected (``invalid_grant``). B leaves rotation; A — a different grant, a different account — keeps its pool row (id, source, first place in the fill_first order), its singleton tokens, and keeps serving turns in later processes.""" @@ -235,7 +230,7 @@ def test_dead_grant_on_independent_login_keeps_login_a(tmp_path, scenario) -> No b_row = next((r for r in pools[0] if r["id"] == "login-b"), None) assert b_row is None or b_row.get("last_status") == "dead", b_row # A is a different grant: nothing about it may change. - with bug_assertions(): + with bug_assertions(KNOWN, "independent_login_survives"): assert stores[0]["providers"]["openai-codex"]["tokens"] == a, stores[0]["providers"]["openai-codex"] for rows in pools: assert _live_order(rows) == [("login-a", "device_code")], ( diff --git a/tests/e2e/core/providers/test_openai_pricing.py b/tests/e2e/core/providers/test_openai_pricing.py index 14562b262b8b0..252de704c96b3 100644 --- a/tests/e2e/core/providers/test_openai_pricing.py +++ b/tests/e2e/core/providers/test_openai_pricing.py @@ -38,22 +38,19 @@ REPO_ROOT, Home, bug_assertions, - known_marks, write_sitecustomize_shim, ) from tests.e2e.core.providers._openai_tui import TuiGateway pytestmark = pytest.mark.skipif(not sys.platform.startswith("linux"), reason="subprocess harness is Linux-gated") -KNOWN: dict[str, str] = { - "custom_twin_priced": "#120757 custom: row pointing at a priced aggregator gets no picker prices", +KNOWN: dict[str, tuple[str, str]] = { + "custom_twin_priced": ( + r"custom:openrouter row has 0/\d+ models priced", + "#120757 custom: row pointing at a priced aggregator gets no picker prices"), } -def known(name: str) -> list: - return known_marks(KNOWN, name) - - VENDOR_ORIGIN = "https://openrouter.ai" # (id, $/token prompt, $/token completion) — the catalog the fake aggregator publishes. CATALOG = [ @@ -254,15 +251,14 @@ def test_relay_without_catalog_gets_no_borrowed_prices(picker, slug: str) -> Non assert _priced(relay) == {}, f"{slug} (upstream: the relay) borrowed prices: {relay.get('pricing')}" -@pytest.mark.parametrize("scenario", [pytest.param("custom_twin_priced", marks=known("custom_twin_priced"))]) -def test_config_defined_twin_row_is_priced_like_the_aggregator(picker, scenario) -> None: +def test_config_defined_twin_row_is_priced_like_the_aggregator(picker) -> None: """The ``custom:openrouter`` row serves the same upstream as the built-in row; its models must carry the aggregator's prices, not none.""" payload, _requests = picker twin, builtin = _row(payload, "custom:openrouter"), _row(payload, "openrouter") assert twin is not None and set(TWIN_MODELS) <= set(twin.get("models") or []), payload.get("providers") priced = _priced(twin) - with bug_assertions(): + with bug_assertions(KNOWN, "custom_twin_priced"): assert set(priced) == set(TWIN_MODELS), ( f"custom:openrouter row has {len(priced)}/{len(TWIN_MODELS)} models priced " f"(pricing={twin.get('pricing')!r})") diff --git a/tests/e2e/core/providers/test_openai_responses.py b/tests/e2e/core/providers/test_openai_responses.py index a1fea2e56d3f3..04d466a0c8e7c 100644 --- a/tests/e2e/core/providers/test_openai_responses.py +++ b/tests/e2e/core/providers/test_openai_responses.py @@ -34,7 +34,6 @@ bug_assertions, db_messages, db_tool_calls, - known_marks, oneshot, responses_input_items, responses_provider_config, @@ -54,17 +53,15 @@ pytestmark = pytest.mark.skipif(not sys.platform.startswith("linux"), reason="subprocess harness is Linux-gated") -# Scenario -> "#issue one-line symptom" for scenarios red on origin/main (strict xfail that -# only a KnownBugError from bug_assertions() satisfies; see known_marks). -KNOWN: dict[str, str] = { - "soft_failure_recovers_on_primary": "#120399 HTTP-200 invalid_encrypted_content skips replay-strip recovery", +# Scenario -> (pattern, "#issue one-line symptom") for scenarios red on origin/main: a +# KnownBugError from bug_assertions() matching the pattern XFAILs the cell (known_gate). +KNOWN: dict[str, tuple[str, str]] = { + "soft_failure_recovers_on_primary": ( + r"(?s)fallback engaged instead of replay recovery: .*stdout='FROM-FALLBACK", + "#120399 HTTP-200 invalid_encrypted_content skips replay-strip recovery"), } -def known(name: str) -> list: - return known_marks(KNOWN, name) - - def _encs(body: dict) -> list: return [i.get("encrypted_content") for i in responses_input_items(body, "reasoning")] @@ -145,8 +142,7 @@ def _stale_blob_session(h: Home, srv: FakeResponsesServer, fallback: FakeLLMServ @pytest.mark.parametrize("rejection", [ pytest.param(HttpError(400, "Encrypted content could not be decrypted or parsed.", code="invalid_encrypted_content", type="invalid_request_error"), id="http_400"), - pytest.param(SoftFail(), id="http_200_soft_failure", - marks=known("soft_failure_recovers_on_primary")), + pytest.param(SoftFail(), id="http_200_soft_failure"), ]) def test_rejected_encrypted_replay_is_stripped_and_primary_retried(tmp_path, rejection) -> None: h = Home(tmp_path) @@ -158,7 +154,8 @@ def test_rejected_encrypted_replay_is_stripped_and_primary_retried(tmp_path, rej fallback_mains = fallback.main_requests() assert _encs(mains[1]) == ["ENC-STALE"], "precondition: the resumed turn replays the stale blob" - with bug_assertions(): + scenario = "soft_failure_recovers_on_primary" if isinstance(rejection, SoftFail) else "http_400" + with bug_assertions(KNOWN, scenario): assert fallback_mains == [], f"fallback engaged instead of replay recovery: {run.describe()}" assert len(mains) == 3, [m.get("input") for m in mains] assert _encs(mains[2]) == [], "the retry must drop the rejected blob" diff --git a/tests/e2e/core/security/_helpers.py b/tests/e2e/core/security/_helpers.py index 85f4f2acb73be..47819e14906ff 100644 --- a/tests/e2e/core/security/_helpers.py +++ b/tests/e2e/core/security/_helpers.py @@ -6,8 +6,8 @@ the boundary's observable outcome: files on disk, state.db rows, logs, and the next wire request. ``BoundaryBreach`` is raised (never a bare ``assert``) when the guarded boundary itself fails, so a -``KNOWN`` strict xfail (``raises=BoundaryBreach``) matches only the tracked bug; a harness failure -(boot, timeout, lost turn) still fails the test loudly. +``KNOWN`` entry gated with ``known_gate(..., raises=BoundaryBreach)`` matches only the tracked bug; a +harness failure (boot, timeout, lost turn) is a plain ``AssertionError`` and still fails the test loudly. """ from __future__ import annotations @@ -24,20 +24,14 @@ from pathlib import Path from typing import Any, Callable, Iterable -import pytest - REPO_ROOT = Path(__file__).resolve().parents[4] class BoundaryBreach(AssertionError): - """A security boundary did not hold (secret leaked, file outside a store touched, approval bypassed).""" - + """A security boundary did not hold (secret leaked, file outside a store touched, approval bypassed). -def known_param(name: str, known: dict[str, str]) -> Any: - """``pytest.param`` for a scenario, marked strict-xfail on ``BoundaryBreach`` when it is KNOWN.""" - if name in known: - return pytest.param(name, marks=pytest.mark.xfail(strict=True, raises=BoundaryBreach, reason=known[name])) - return name + Raised only at a boundary assertion: it is the type ``known_failure`` / ``known_gate`` accept + (``raises=BoundaryBreach``) for a KNOWN gap, so a harness failure can never be absorbed.""" def canary(label: str) -> str: diff --git a/tests/e2e/core/security/_redact.py b/tests/e2e/core/security/_redact.py index d7ae7725f66b7..af4bac5f275c6 100644 --- a/tests/e2e/core/security/_redact.py +++ b/tests/e2e/core/security/_redact.py @@ -15,7 +15,7 @@ import sys from dataclasses import dataclass, field from pathlib import Path -from typing import Any, Callable +from typing import Any, Callable, Mapping import pytest @@ -253,20 +253,23 @@ class World: runs: dict[str, str] -def cells(known: dict[str, str], *, platform: bool) -> list[Any]: +def cell_id(scenario: str, sink: str) -> str: + """Test id and KNOWN key of one (scenario, sink) cell.""" + return f"{scenario}-{SINK_IDS[sink]}" + + +def cells(known: Mapping[str, Any], *, platform: bool) -> list[Any]: """One ``pytest.param(scenario, sink)`` per sink a scenario is held to (``platform``: the surface has a - platform wire). A KNOWN key is a cell id ``-`` and strict-xfails ONLY that sink, so a - new leak of the same scenario into any other sink is a plain red, never absorbed by the known gap.""" + platform wire). A KNOWN key is a cell id (:func:`cell_id`) and gates ONLY that sink (``known_gate``), so + a new leak of the same scenario into any other sink is a plain red, never absorbed by the known gap.""" out, ids = [], set() for name, scenario in SCENARIOS.items(): for sink in scenario.sinks: if sink == PLATFORM and not platform: continue - cid = f"{name}-{SINK_IDS[sink]}" + cid = cell_id(name, sink) ids.add(cid) - marks = ([pytest.mark.xfail(strict=True, raises=BoundaryBreach, reason=known[cid])] - if cid in known else []) - out.append(pytest.param(name, sink, id=cid, marks=marks)) + out.append(pytest.param(name, sink, id=cid)) stale = sorted(set(known) - ids) assert not stale, f"KNOWN names cells that do not exist: {stale}" return out diff --git a/tests/e2e/core/security/_traversal.py b/tests/e2e/core/security/_traversal.py index 5022dc02ce567..e42a7132bdb78 100644 --- a/tests/e2e/core/security/_traversal.py +++ b/tests/e2e/core/security/_traversal.py @@ -36,7 +36,7 @@ def run_tool_calls(home: Path, calls: list[tuple[str, dict[str, Any]]], *, cwd: """Run one ``hermes chat -q`` turn whose model issues ``calls`` sequentially; return each tool result text, in call order. ``prepare`` runs after the home (config.yaml + .env) is written and before Hermes starts (snapshot pre-run state there). Harness failures (non-zero exit, a lost tool - result) are plain ``AssertionError`` so a KNOWN strict xfail never masks them.""" + result) are plain ``AssertionError`` so a KNOWN ``known_gate`` (``raises=BoundaryBreach``) never masks them.""" key = H.canary("sk-traversal") script: list[Any] = [ToolCall(name, args) for name, args in calls] + [Text("done")] with FakeLLMServer(script, api_key=key) as srv: diff --git a/tests/e2e/core/security/test_approval_boundary.py b/tests/e2e/core/security/test_approval_boundary.py index 7b1bb16548c3c..864ee06bcea34 100644 --- a/tests/e2e/core/security/test_approval_boundary.py +++ b/tests/e2e/core/security/test_approval_boundary.py @@ -31,7 +31,8 @@ import pytest from tests.e2e.core.delivery._fake_platform import GatewayProcess, wait_until -from tests.e2e.core.security._helpers import BoundaryBreach, known_param, run_hermes, write_home +from tests.e2e.core._pending_fixes import known_gate +from tests.e2e.core.security._helpers import BoundaryBreach, run_hermes, write_home from tests.fakes.fake_llm_provider import FakeLLMServer, Text, ToolCall pytestmark = pytest.mark.skipif(sys.platform == "win32", reason="POSIX shell variants + process-group harness") @@ -67,8 +68,9 @@ "python_c": "python3 -c \"import shutil; shutil.rmtree('{X}')\"", } -# Detector misses confirmed live on origin/main (strict xfail on BoundaryBreach). None today. -KNOWN: Dict[str, str] = {} +# Detector misses confirmed live on origin/main: variant -> (pattern, "#issue symptom"), gated with +# ``known_gate(..., raises=BoundaryBreach)``. None today. +KNOWN: Dict[str, tuple[str, str]] = {} CONTROL = "touch benign_marker" @@ -154,14 +156,15 @@ def test_cli_benign_control_runs(cli_deny): assert (cli_deny["victims"] / "benign_marker").exists(), "benign control command did not run" -@pytest.mark.parametrize("variant", [known_param(n, KNOWN) for n in VARIANTS]) +@pytest.mark.parametrize("variant", list(VARIANTS)) def test_cli_dangerous_variant_blocked_without_approval(cli_deny, variant): victim = cli_deny["victims"] / variant result = cli_deny["results"][VARIANTS[variant].format(X=variant)] - if not (victim / "keep.txt").exists(): - raise BoundaryBreach(f"{variant!r} deleted its victim with nobody to approve it; tool result: {result}") - if result.get("exit_code") == 0 or result.get("status") != "blocked": - raise BoundaryBreach(f"{variant!r}: tool result not marked blocked: {result}") + with known_gate(KNOWN, variant, raises=BoundaryBreach): + if not (victim / "keep.txt").exists(): + raise BoundaryBreach(f"{variant!r} deleted its victim with nobody to approve it; tool result: {result}") + if result.get("exit_code") == 0 or result.get("status") != "blocked": + raise BoundaryBreach(f"{variant!r}: tool result not marked blocked: {result}") # --- Gateway -------------------------------------------------------------------------------------- diff --git a/tests/e2e/core/security/test_attach_handshake.py b/tests/e2e/core/security/test_attach_handshake.py index d4a11e4f1e5cc..c8a785100f3ad 100644 --- a/tests/e2e/core/security/test_attach_handshake.py +++ b/tests/e2e/core/security/test_attach_handshake.py @@ -47,12 +47,12 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.security._helpers import ( BoundaryBreach, canary, hermetic_env, kill_group, - known_param, run_hermes, run_python, write_home, @@ -63,8 +63,10 @@ pytest.mark.skipif(shutil.which("node") is None, reason="the TUI hand-off needs node on PATH"), ] -KNOWN: dict[str, str] = { - "impersonator": "#120604 attach handshake accepts an echoing loopback listener", +KNOWN: dict[str, tuple[str, str]] = { + "impersonator": (r"^impersonator: TUI was attached to the rogue listener's socket: " + r"ws://127\.0\.0\.1:\d+/api/ws\?token=stolen-ws-token-", + "#120604 attach handshake accepts an echoing loopback listener"), } SCENARIOS = ["impersonator", "foreign_owner"] @@ -257,7 +259,7 @@ def _describe(result: subprocess.CompletedProcess) -> str: return f"rc={result.returncode}\nstdout:\n{result.stdout[-1500:]}\nstderr:\n{result.stderr[-2500:]}" -@pytest.mark.parametrize("scenario", [known_param(n, KNOWN) for n in SCENARIOS]) +@pytest.mark.parametrize("scenario", SCENARIOS) def test_rogue_listener_is_not_attached(home: Path, scenario: str) -> None: sid = f"attach-{uuid.uuid4().hex[:12]}" stolen = canary("stolen-ws-token") @@ -275,11 +277,12 @@ def test_rogue_listener_is_not_attached(home: Path, scenario: str) -> None: assert (result.returncode == 0) == (handoff is not None), f"inconsistent launch outcome\n{_describe(result)}" gateway = (handoff or {}).get("gateway_url") or "" - if stolen in gateway or gateway.startswith(listener.origin.replace("http:", "ws:")): - raise BoundaryBreach(f"{scenario}: TUI was attached to the rogue listener's socket: {gateway}") leaked = [name for name, text in (("stdout", result.stdout), ("stderr", result.stderr)) if stolen in text] - if leaked: - raise BoundaryBreach(f"{scenario}: the rogue listener's token was echoed to {leaked}") + with known_gate(KNOWN, scenario, raises=BoundaryBreach): + if stolen in gateway or gateway.startswith(listener.origin.replace("http:", "ws:")): + raise BoundaryBreach(f"{scenario}: TUI was attached to the rogue listener's socket: {gateway}") + if leaked: + raise BoundaryBreach(f"{scenario}: the rogue listener's token was echoed to {leaked}") def test_genuine_owner_is_attached(home: Path) -> None: diff --git a/tests/e2e/core/security/test_profile_secret_scope.py b/tests/e2e/core/security/test_profile_secret_scope.py index 2dbf10a66f202..ab54773284a08 100644 --- a/tests/e2e/core/security/test_profile_secret_scope.py +++ b/tests/e2e/core/security/test_profile_secret_scope.py @@ -36,17 +36,21 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.tenancy import _helpers as T from . import _scope as S -from ._helpers import BoundaryBreach, known_param, poll +from ._helpers import BoundaryBreach, poll pytestmark = pytest.mark.skipif(sys.platform == "win32", reason="POSIX process groups (serve teardown)") PROFILES = ("default", "beta") -KNOWN: dict[str, str] = { - "plugin_route": "#120310 /api/plugins// routes run with no profile secret scope under multiplex", +KNOWN: dict[str, tuple[str, str]] = { + # every read fails closed (no scope bound); a route that borrows another profile's value stays red + "plugin_route": (r"^plugin_route: plugin API route did not run in the requesting profile's secret scope:" + r"(\n \w+: get_secret raised UnscopedSecretError)+\Z", + "#120310 /api/plugins// routes run with no profile secret scope under multiplex"), } @@ -127,7 +131,7 @@ def test_scoped_sibling_route_resolves_each_profiles_secret(host: Host) -> None: f"(own ends {own[-4:]!r}; matches {others or 'nobody'})") -@pytest.mark.parametrize("scenario", [known_param("plugin_route", KNOWN)]) +@pytest.mark.parametrize("scenario", ["plugin_route"]) def test_plugin_route_runs_in_requesting_profile_scope(host: Host, scenario: str) -> None: replies: list[str] = [] for name in ("default", "beta", "default"): @@ -140,9 +144,10 @@ def test_plugin_route_runs_in_requesting_profile_scope(host: Host, scenario: str if value != host.secrets[name]: owner = next((u for u, v in host.secrets.items() if v == value), "") replies.append(f"{name}: got {owner}'s value {value!r}") - if replies: - raise BoundaryBreach(f"{scenario}: plugin API route did not run in the requesting profile's " - "secret scope:\n " + "\n ".join(replies)) + with known_gate(KNOWN, scenario, raises=BoundaryBreach): + if replies: + raise BoundaryBreach(f"{scenario}: plugin API route did not run in the requesting profile's " + "secret scope:\n " + "\n ".join(replies)) def test_mcp_subprocess_env_holds_only_own_profile_secrets(host: Host) -> None: diff --git a/tests/e2e/core/security/test_secret_redaction.py b/tests/e2e/core/security/test_secret_redaction.py index 631cf75486fae..09084a81b2e04 100644 --- a/tests/e2e/core/security/test_secret_redaction.py +++ b/tests/e2e/core/security/test_secret_redaction.py @@ -25,16 +25,17 @@ import pytest -from tests.e2e.core.security._helpers import run_hermes, write_home +from tests.e2e.core._pending_fixes import known_gate +from tests.e2e.core.security._helpers import BoundaryBreach, run_hermes, write_home from tests.e2e.core.security._redact import ( - CONFIG, SCENARIOS, Ctx, Director, Secrets, World, assert_harness_sane, cells, check, collect, echo_preconditions, - prompt_for, seed_workspace, + CONFIG, SCENARIOS, Ctx, Director, Secrets, World, assert_harness_sane, cell_id, cells, check, collect, + echo_preconditions, prompt_for, seed_workspace, ) from tests.fakes.fake_llm_provider import FakeLLMServer pytestmark = pytest.mark.skipif(sys.platform == "win32", reason="POSIX shell commands (cat | tee, curl)") -KNOWN: dict[str, str] = {} # cell id ``-`` -> "#issue symptom" +KNOWN: dict[str, tuple[str, str]] = {} # cell id ``-`` -> (pattern, "#issue symptom") _SESSION_RE = re.compile(r"session_id:\s*(\S+)") @@ -69,4 +70,5 @@ def cli_world(tmp_path_factory) -> World: @pytest.mark.parametrize("scenario, sink", cells(KNOWN, platform=False)) def test_cli_turn_never_persists_or_replays_a_secret(cli_world: World, scenario: str, sink: str) -> None: - check(cli_world, scenario, sink) + with known_gate(KNOWN, cell_id(scenario, sink), raises=BoundaryBreach): + check(cli_world, scenario, sink) diff --git a/tests/e2e/core/security/test_secret_redaction_gateway.py b/tests/e2e/core/security/test_secret_redaction_gateway.py index 279e8232b4af6..d1aff851a3885 100644 --- a/tests/e2e/core/security/test_secret_redaction_gateway.py +++ b/tests/e2e/core/security/test_secret_redaction_gateway.py @@ -13,10 +13,11 @@ import pytest -from tests.e2e.core.security._helpers import write_home +from tests.e2e.core._pending_fixes import known_gate +from tests.e2e.core.security._helpers import BoundaryBreach, write_home from tests.e2e.core.security._redact import ( - CONFIG, SCENARIOS, Ctx, Director, LoggingGateway, Secrets, World, assert_harness_sane, cells, chat_for, check, - collect, echo_preconditions, prompt_for, seed_workspace, + CONFIG, SCENARIOS, Ctx, Director, LoggingGateway, Secrets, World, assert_harness_sane, cell_id, cells, chat_for, + check, collect, echo_preconditions, prompt_for, seed_workspace, ) from tests.e2e.core.delivery._fake_platform import wait_until from tests.fakes.fake_llm_provider import FakeLLMServer @@ -25,8 +26,10 @@ # Keyed by cell id ``-``: only the platform wire of the streamed answer is the known gap; # the same answer reaching gateway.log, state.db, an export or the next request stays a plain red. -KNOWN: dict[str, str] = { - "assistant_text-platform": "#56039 streamed gateway replies reach the platform without secret redaction", +KNOWN: dict[str, tuple[str, str]] = { + "assistant_text-platform": ( + r"^assistant_text: secret reached the redacted sink 'platform wire':\n platform wire \[chat c-assistant_text\]", + "#56039 streamed gateway replies reach the platform without secret redaction"), } PLATFORM = "fk_tg" @@ -70,4 +73,5 @@ def gw_world(tmp_path_factory) -> World: @pytest.mark.parametrize("scenario, sink", cells(KNOWN, platform=True)) def test_gateway_turn_never_persists_delivers_or_replays_a_secret(gw_world: World, scenario: str, sink: str) -> None: - check(gw_world, scenario, sink) + with known_gate(KNOWN, cell_id(scenario, sink), raises=BoundaryBreach): + check(gw_world, scenario, sink) diff --git a/tests/e2e/core/security/test_store_traversal.py b/tests/e2e/core/security/test_store_traversal.py index c9400b897adf7..4f1af9ae1ab63 100644 --- a/tests/e2e/core/security/test_store_traversal.py +++ b/tests/e2e/core/security/test_store_traversal.py @@ -19,6 +19,7 @@ from __future__ import annotations import json +import re import sys from dataclasses import dataclass, field from pathlib import Path @@ -26,6 +27,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.security import _helpers as H from tests.e2e.core.security._traversal import digest, result_json, tool_text from tests.e2e.core.tenancy._helpers import TuiBackend @@ -51,7 +53,24 @@ "pinned_by_category": ("research/my-skill", "research/my-skill"), "essential_by_category": ("autonomous-ai-agents/hermes-agent", "autonomous-ai-agents/hermes-agent"), } -KNOWN: dict[str, str] = {**dict.fromkeys(TRAVERSALS, _ISSUE_STORE), **dict.fromkeys(DELETES, _ISSUE_DELETE)} + + +def _store_signature(command: str, where: str) -> str: + """The traversal reaching its victim: deleted/rewritten (reject, approve) or its content shown (diff).""" + if where == "home": + spelled, victim = re.escape("../../auth"), "auth.json" + else: + spelled, victim = r"/[^']+/outside/victim", "victim.json" + return (rf"^/{command} '{spelled}': " + rf"(?:{re.escape(victim)} was deleted or rewritten|the command output carries the victim's content)") + + +# scenario -> (pattern, "#issue symptom"), gated with ``known_gate(..., raises=H.BoundaryBreach)`` +KNOWN: dict[str, tuple[str, str]] = { + **{n: (_store_signature(*TRAVERSALS[n]), _ISSUE_STORE) for n in TRAVERSALS}, + **{n: (rf"^skill_manage delete {re.escape(repr(name))} removed skills/{re.escape(rel)}: \{{'success': True", + _ISSUE_DELETE) for n, (name, rel) in DELETES.items()}, +} _STORE_CONFIG = "memory:\n memory_enabled: true\n write_approval: true\nskills:\n write_approval: true\n" _SKILL = "---\nname: {name}\ndescription: {desc}\n---\n\n# {name}\n\n{body}\n" @@ -137,7 +156,7 @@ def _plant_victim(s: Store, where: str) -> tuple[Path, str, str]: return victim, spelled, mark -@pytest.mark.parametrize("scenario", [H.known_param(n, KNOWN) for n in TRAVERSALS]) +@pytest.mark.parametrize("scenario", list(TRAVERSALS)) def test_pending_id_never_leaves_the_store(store: Store, scenario: str) -> None: command, where = TRAVERSALS[scenario] victim, spelled, mark = _plant_victim(store, where) @@ -153,8 +172,9 @@ def test_pending_id_never_leaves_the_store(store: Store, scenario: str) -> None: (mark in store.memory_blob(), "the victim's payload was applied to the memory store"), (pending_after != pending_before, f"real staged records changed: {pending_before} -> {pending_after}"), ) - if breaches := [msg for failed, msg in checks if failed]: - raise H.BoundaryBreach(f"/{command} {spelled!r}: " + "; ".join(breaches) + f"\noutput: {out[:400]}") + with known_gate(KNOWN, scenario, raises=H.BoundaryBreach): + if breaches := [msg for failed, msg in checks if failed]: + raise H.BoundaryBreach(f"/{command} {spelled!r}: " + "; ".join(breaches) + f"\noutput: {out[:400]}") def test_control_valid_reject_drops_only_that_record(store: Store) -> None: @@ -226,11 +246,12 @@ def deletes(tmp_path_factory: pytest.TempPathFactory) -> Iterator[DeleteRun]: yield run -@pytest.mark.parametrize("scenario", [H.known_param(n, KNOWN) for n in DELETES]) +@pytest.mark.parametrize("scenario", list(DELETES)) def test_delete_refuses_pinned_and_essential_by_category(deletes: DeleteRun, scenario: str) -> None: name, rel = DELETES[scenario] - if not deletes.survived[scenario]: - raise H.BoundaryBreach(f"skill_manage delete {name!r} removed skills/{rel}: {deletes.results[scenario]}") + with known_gate(KNOWN, scenario, raises=H.BoundaryBreach): + if not deletes.survived[scenario]: + raise H.BoundaryBreach(f"skill_manage delete {name!r} removed skills/{rel}: {deletes.results[scenario]}") assert deletes.results[scenario].get("success") is False, deletes.results[scenario] diff --git a/tests/e2e/core/security/test_workspace_escape.py b/tests/e2e/core/security/test_workspace_escape.py index ef6729c5b7c73..4536e55d02f77 100644 --- a/tests/e2e/core/security/test_workspace_escape.py +++ b/tests/e2e/core/security/test_workspace_escape.py @@ -25,6 +25,7 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.security import _helpers as H from tests.e2e.core.security._traversal import digest, result_json, run_tool_calls @@ -42,8 +43,17 @@ "ssh_config_symlink_parent_gated", "read_auth_symlink", "read_webhooks_dotdot", "read_google_oauth_symlink_dir", "read_project_env_symlink", "search_through_symlink_dir") -KNOWN: dict[str, str] = dict.fromkeys( - QUARANTINE_SCENARIOS, "#121278 read denylist misses auth.json.corrupt, the quarantined copy of a corrupt auth store") +_ISSUE_QUARANTINE = "#121278 read denylist misses auth.json.corrupt, the quarantined copy of a corrupt auth store" +# scenario -> (pattern, "#issue symptom"), gated with ``known_gate(..., raises=H.BoundaryBreach)`` in every +# boundary test below, so a future entry for any spelling is merge-order safe. +KNOWN: dict[str, tuple[str, str]] = { + "read_quarantined_auth_copy": ( + r"^read_quarantined_auth_copy: read_file \{'path': '[^']*auth\.json\.corrupt'\}: " + r"the tool result carries the protected file's content", _ISSUE_QUARANTINE), + "search_quarantined_auth_copy": ( + r"^search_quarantined_auth_copy: search_files \{[^}]*\}: the tool result carries the protected file's " + r'content\nresult: .*"path": "[^"]*auth\.json\.corrupt"', _ISSUE_QUARANTINE), +} @dataclass @@ -260,19 +270,22 @@ def _check(run: Run, scenario: str) -> None: f"{scenario}: the boundary held on disk but the tool did not report a refusal: {text[:400]}" -@pytest.mark.parametrize("scenario", [H.known_param(n, KNOWN) for n in CONFINED_SCENARIOS]) +@pytest.mark.parametrize("scenario", CONFINED_SCENARIOS) def test_write_sandbox_holds_under_every_spelling(confined: Run, scenario: str) -> None: - _check(confined, scenario) + with known_gate(KNOWN, scenario, raises=H.BoundaryBreach): + _check(confined, scenario) -@pytest.mark.parametrize("scenario", [H.known_param(n, KNOWN) for n in PROTECTED_SCENARIOS]) +@pytest.mark.parametrize("scenario", PROTECTED_SCENARIOS) def test_protected_paths_hold_under_every_spelling(protected: Run, scenario: str) -> None: - _check(protected, scenario) + with known_gate(KNOWN, scenario, raises=H.BoundaryBreach): + _check(protected, scenario) -@pytest.mark.parametrize("scenario", [H.known_param(n, KNOWN) for n in QUARANTINE_SCENARIOS]) +@pytest.mark.parametrize("scenario", QUARANTINE_SCENARIOS) def test_quarantined_auth_store_stays_read_denied(confined: Run, scenario: str) -> None: - _check(confined, scenario) + with known_gate(KNOWN, scenario, raises=H.BoundaryBreach): + _check(confined, scenario) def test_control_sandbox_allows_writes_that_resolve_inside(confined: Run) -> None: diff --git a/tests/e2e/core/upgrade/test_update_userstate_import.py b/tests/e2e/core/upgrade/test_update_userstate_import.py index c861e194d4c6d..7313dd68e6d5d 100644 --- a/tests/e2e/core/upgrade/test_update_userstate_import.py +++ b/tests/e2e/core/upgrade/test_update_userstate_import.py @@ -31,14 +31,7 @@ class Gap(Exception): - """Contract breach tracked in KNOWN (not an AssertionError: a crash still fails the test).""" - - -KNOWN: dict[str, str] = {} - - -def known(key: str): - return pytest.mark.xfail(strict=True, raises=Gap, reason=KNOWN[key]) if key in KNOWN else () + """Contract breach (not an AssertionError, so a crash reads differently from a wrong answer).""" def _home(root: Path, name: str) -> dict: @@ -96,7 +89,7 @@ def _rotten_member(src: Path, dst: Path) -> None: BROKEN = {"truncated": _truncated, "not-a-zip": _not_a_zip, "rotten-member": _rotten_member} -@pytest.mark.parametrize("kind", [pytest.param(k, marks=known(k)) for k in BROKEN]) +@pytest.mark.parametrize("kind", list(BROKEN)) def test_import_of_a_broken_archive_fails_and_leaves_the_home_alone(archive, kind): root, good = archive bad = root / f"{kind}.zip" @@ -117,8 +110,7 @@ def test_import_of_a_broken_archive_fails_and_leaves_the_home_alone(archive, kin assert not skill.exists() or skill.read_text(encoding="utf-8") == SKILL, "a corrupt member was written as the skill" -@pytest.mark.parametrize("key", [pytest.param("partial", marks=known("partial"))]) -def test_import_that_skips_members_reports_incomplete(archive, key): +def test_import_that_skips_members_reports_incomplete(archive): root, good = archive env = _home(root, "dst-partial") hh = Path(env["HERMES_HOME"]) diff --git a/tests/e2e/core/windows/_helpers.py b/tests/e2e/core/windows/_helpers.py index 1984dea571130..d2b4d0d05d9cd 100644 --- a/tests/e2e/core/windows/_helpers.py +++ b/tests/e2e/core/windows/_helpers.py @@ -324,32 +324,11 @@ def nonce(prefix: str) -> str: class KnownBugSymptom(Exception): - """Raised ONLY at the assertion that pins a tracked bug's symptom. Not an AssertionError: - a KNOWN entry's strict xfail matches just this, so a harness failure or any other broken - invariant in the same test still fails loudly instead of hiding behind the xfail.""" + """Raised ONLY at the assertion that pins a tracked bug's symptom. Not an AssertionError: a + file's ``KNOWN`` gate (``known_gate(KNOWN, key, raises=KnownBugSymptom)``) accepts just this, + so a harness failure or any other broken invariant in the same test still fails loudly.""" def expect(ok: bool, message: str) -> None: if not ok: raise KnownBugSymptom(message) - - -def known_marks(name: str, table: dict[str, str]) -> list[Any]: - """Marks for a test named in a file's ``KNOWN`` table: strict xfail on the symptom only. - Strict, so the test turns red the moment the bug is fixed and the entry must go.""" - import pytest - - if name not in table: - return [] - return [pytest.mark.xfail(strict=True, raises=KnownBugSymptom, reason=table[name])] - - -def known(name: str, table: dict[str, str]) -> Callable[[Callable], Callable]: - """Decorator form of :func:`known_marks` for a non-parametrized test.""" - - def apply(fn: Callable) -> Callable: - for mark in known_marks(name, table): - fn = mark(fn) - return fn - - return apply diff --git a/tests/e2e/core/windows/test_cron_scripts.py b/tests/e2e/core/windows/test_cron_scripts.py index d588d3f04322d..c9aaebcba475f 100644 --- a/tests/e2e/core/windows/test_cron_scripts.py +++ b/tests/e2e/core/windows/test_cron_scripts.py @@ -21,12 +21,15 @@ import pytest -from tests.e2e.core.windows._helpers import WinHome, expect, hermes, known, make_home, nonce +from tests.e2e.core._pending_fixes import known_gate +from tests.e2e.core.windows._helpers import KnownBugSymptom, WinHome, expect, hermes, make_home, nonce pytestmark = [pytest.mark.windows_only, pytest.mark.integration] -KNOWN: dict[str, str] = { - "sh_script": "#120504 cron .sh scripts resolve bash via bare PATH lookup (WSL stub), not Git Bash", +# key -> (the bug's own failure signature, "#issue reason"); see _pending_fixes.known_failure. +KNOWN: dict[str, tuple[str, str]] = { + "sh_script": (r"^bare PATH lookup finds (None|'[^']*(?i:system32|windowsapps)[^']*'); job status", + "#120504 cron .sh scripts resolve bash via bare PATH lookup (WSL stub), not Git Bash"), } _JOB_ID = re.compile(r"Created job: (\S+)") @@ -75,7 +78,6 @@ def test_python_script_job_delivers_stdout(tmp_path: Path) -> None: assert f"{marker} win32" in output, f"script stdout not delivered:\n{output}" -@known("sh_script", KNOWN) def test_sh_script_job_runs_under_git_bash(tmp_path: Path) -> None: git_bash = Path(os.environ.get("ProgramFiles", r"C:\Program Files"), "Git", "bin", "bash.exe") assert git_bash.is_file(), f"precondition: Git for Windows installed at {git_bash}" @@ -87,6 +89,7 @@ def test_sh_script_job_runs_under_git_bash(tmp_path: Path) -> None: native_path = _native_process_path() job, output = _run_script_job(home, "report.sh", env_extra={"PATH": native_path}) bare = shutil.which("bash", path=native_path) - expect(job["last_status"] == "ok" and marker in output, - f"bare PATH lookup finds {bare!r}; job status {job['last_status']!r}: {job.get('last_error')}\n{output}") + with known_gate(KNOWN, "sh_script", raises=KnownBugSymptom): + expect(job["last_status"] == "ok" and marker in output, + f"bare PATH lookup finds {bare!r}; job status {job['last_status']!r}: {job.get('last_error')}\n{output}") assert "_NT-" in output, f".sh job ran, but not under Git Bash (MSYS/MinGW uname):\n{output}" diff --git a/tests/e2e/core/windows/test_prompt_paths.py b/tests/e2e/core/windows/test_prompt_paths.py index 410184fac9d2d..a14c02f22f60e 100644 --- a/tests/e2e/core/windows/test_prompt_paths.py +++ b/tests/e2e/core/windows/test_prompt_paths.py @@ -18,11 +18,11 @@ import pytest +from tests.e2e.core._pending_fixes import known_gate from tests.e2e.core.windows._helpers import ( + KnownBugSymptom, expect, hermes, - known_marks, - known, last_user, make_home, nonce, @@ -34,10 +34,14 @@ pytestmark = [pytest.mark.windows_only, pytest.mark.integration] -KNOWN: dict[str, str] = { - "native-backslash": "#121150 subdirectory hints never load from native Windows paths (POSIX shlex)", - "chain_labels": "#121015 AGENTS.md chain labels use os.path.relpath separators (..\\AGENTS.md)", - "folder_header": "#121114 @folder listing header mixes separators on Windows (pkg\\sub/)", +# key -> (the bug's own failure signature, "#issue reason"); see _pending_fixes.known_failure. +KNOWN: dict[str, tuple[str, str]] = { + "native-backslash": (r"^native-backslash: backend/AGENTS\.md never reached the model after ", + "#121150 subdirectory hints never load from native Windows paths (POSIX shlex)"), + "chain_labels": (r"^AGENTS\.md chain headings are not the portable spelling: \[.*\\", + "#121015 AGENTS.md chain labels use os.path.relpath separators (..\\AGENTS.md)"), + "folder_header": (r"^@folder header lines: \['pkg\\+sub/'\]", + "#121114 @folder listing header mixes separators on Windows (pkg\\sub/)"), } # spelling -> terminal command the model issues (the issue's own repro commands) @@ -47,9 +51,7 @@ } -@pytest.mark.parametrize("spelling", [ - pytest.param(name, marks=known_marks(name, KNOWN)) for name in HINT_COMMANDS -]) +@pytest.mark.parametrize("spelling", list(HINT_COMMANDS)) def test_subdirectory_hint_reaches_model(spelling: str, tmp_path: Path) -> None: canary = nonce("BACKEND-RULES") with FakeLLMServer([ToolCall("terminal", {"command": HINT_COMMANDS[spelling]}), Text("done")]) as srv: @@ -61,12 +63,12 @@ def test_subdirectory_hint_reaches_model(spelling: str, tmp_path: Path) -> None: assert res.returncode == 0, res.tail() results = tool_results(srv) assert len(results) == 1, f"expected one terminal result on the wire, got {results}" - expect(canary in results[0], - f"{spelling}: backend/AGENTS.md never reached the model after {HINT_COMMANDS[spelling]!r}:\n" - f"{results[0][-1500:]}") + with known_gate(KNOWN, spelling, raises=KnownBugSymptom): + expect(canary in results[0], + f"{spelling}: backend/AGENTS.md never reached the model after {HINT_COMMANDS[spelling]!r}:\n" + f"{results[0][-1500:]}") -@known("chain_labels", KNOWN) def test_agents_chain_labels_are_os_independent(tmp_path: Path) -> None: canaries = {name: nonce(name.upper()) for name in ("root", "pkg", "inner")} with FakeLLMServer([Text("done")]) as srv: @@ -83,11 +85,11 @@ def test_agents_chain_labels_are_os_independent(tmp_path: Path) -> None: missing = [name for name, c in canaries.items() if c not in prompt] assert not missing, f"AGENTS.md chain members missing from the system prompt: {missing}" headings = [line[3:] for line in prompt.splitlines() if line.startswith("## ") and "AGENTS.md" in line] - expect(headings == ["../../AGENTS.md", "../AGENTS.md", "AGENTS.md"], - f"AGENTS.md chain headings are not the portable spelling: {headings}") + with known_gate(KNOWN, "chain_labels", raises=KnownBugSymptom): + expect(headings == ["../../AGENTS.md", "../AGENTS.md", "AGENTS.md"], + f"AGENTS.md chain headings are not the portable spelling: {headings}") -@known("folder_header", KNOWN) def test_folder_reference_header_uses_one_separator(tmp_path: Path) -> None: body = nonce("FOLDER-FILE") with FakeLLMServer([Text("done")]) as srv: @@ -104,4 +106,5 @@ def test_folder_reference_header_uses_one_separator(tmp_path: Path) -> None: assert "- a.py" in user, f"@folder listing never reached the model:\n{user[-2000:]}" # The header is the one non-entry line naming the folder; entries are "- name" lines. header = [ln.strip() for ln in user.splitlines() if ln.strip().endswith("sub/") and not ln.lstrip().startswith("-")] - expect(header == ["pkg/sub/"], f"@folder header lines: {header}\n{user[-1500:]}") + with known_gate(KNOWN, "folder_header", raises=KnownBugSymptom): + expect(header == ["pkg/sub/"], f"@folder header lines: {header}\n{user[-1500:]}") diff --git a/tests/e2e/core/windows/test_unicode_input.py b/tests/e2e/core/windows/test_unicode_input.py index 7228e6854bbfe..867a5f719c9e2 100644 --- a/tests/e2e/core/windows/test_unicode_input.py +++ b/tests/e2e/core/windows/test_unicode_input.py @@ -36,8 +36,6 @@ pytestmark = [pytest.mark.windows_only, pytest.mark.integration] -KNOWN: dict[str, str] = {} # nothing red on origin/main in this file - TEXT = "Grüße, 日本語 und Emoji 😂👍🏽" REPLY = "Réponse ✓ 😂" BMP = "Grüße, 日本語"