Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 9 additions & 1 deletion tests/e2e/core/_pending_fixes.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@

import contextlib
import re
from typing import Iterator, Tuple, Type
from typing import ContextManager, Iterator, Mapping, Tuple, Type

import pytest

Expand All @@ -30,3 +30,11 @@ def known_failure(pattern: str, reason: str,
if not re.search(pattern, str(exc)):
raise
pytest.xfail(f"{reason} [observed: {str(exc).splitlines()[0][:240]}]")


def known_gate(known: Mapping[str, Tuple[str, str]], key: str,
raises: Type[BaseException] | Tuple[Type[BaseException], ...] = AssertionError) -> ContextManager[None]:
""":func:`known_failure` for a cell a file's ``KNOWN`` table (key -> ``(pattern, reason)``)
names, a no-op for every other cell, so one wrapped block serves gated and plain cells."""
entry = known.get(key)
return known_failure(*entry, raises=raises) if entry else contextlib.nullcontext()
13 changes: 4 additions & 9 deletions tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py
Original file line number Diff line number Diff line change
Expand Up @@ -290,9 +290,9 @@ def _assert_heartbeat(scn: Scenario, hb: Heartbeat, stats: dict[str, Any], turn_

class ToolOutlivedGateway(Exception):
"""The in-flight tool outlived the gateway's exit: its process tree survives and/or its
tool_call was left with no result in state.db. Deliberately NOT an AssertionError: the
known-bug xfail matches only this, so an RPC failure, a crash or any other broken invariant
(heartbeat, exit deadline, integrity) still fails the test."""
tool_call was left with no result in state.db. Deliberately NOT an AssertionError: a
``known_failure(..., raises=ToolOutlivedGateway)`` gate for this bug accepts only it, so an RPC
failure, a crash or any other broken invariant (heartbeat, exit deadline, integrity) still fails."""


def _exit_mid_turn(scn: Scenario, gw: TuiGatewayProcess, hb: Heartbeat, tag: str,
Expand Down Expand Up @@ -456,12 +456,7 @@ def scenario_futures(request: pytest.FixtureRequest, tmp_path_factory: pytest.Te
fut.cancel()


KNOWN_BUGS: dict = {}


@pytest.mark.parametrize("scn", [
pytest.param(s, id=s.id, marks=[KNOWN_BUGS[s.id]] if s.id in KNOWN_BUGS else [])
for s in SCENARIOS])
@pytest.mark.parametrize("scn", [pytest.param(s, id=s.id) for s in SCENARIOS])
def test_tui_gateway_turn_stays_live_under_fault(scn: Scenario, scenario_futures) -> None:
stats = scenario_futures[scn.id].result(timeout=WARMUP_DEADLINE_S + 3 * TURN_DEADLINE_S + 120)
print(json.dumps(stats))
14 changes: 4 additions & 10 deletions tests/e2e/core/dashboard/_issue_helpers.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,8 +8,10 @@
* ``GatewayApiServer``: a real ``python -m gateway.run`` with the API-server platform enabled via
the profile ``.env`` exactly as a user configures it (``API_SERVER_ENABLED``/``_KEY``/``_HOST``/
``_PORT``), retrying the pick-then-bind port race like the parity lane's driver.
* ``KnownIssue`` assertion subclasses: a strict xfail cell declares ``raises=<its subclass>`` so
only the bug's own signature is excused; a boot failure or any other assertion stays red.
* ``KnownIssue`` assertion subclasses: a KNOWN cell gates its final check with
``known_gate(KNOWN, name, raises=<its subclass>)`` (tests/e2e/core/_pending_fixes.py), so only the
bug's own signature is excused; a boot failure or any other assertion stays red, and the cell
simply passes once the fix lands.
"""

from __future__ import annotations
Expand Down Expand Up @@ -48,14 +50,6 @@ class Issue120937(KnownIssue):
"""#120937: /api/sessions/{id}/chat silently truncates a long message."""


def xfail_known(known: dict[str, tuple[str, type[KnownIssue]]], name: str):
"""``pytest.mark.xfail(strict=True)`` for a KNOWN entry, excusing ONLY its issue's exception."""
import pytest

reason, exc = known[name]
return pytest.mark.xfail(strict=True, raises=exc, reason=reason)


# Dashboard with a TTY stdin -----------------------------------------------------------------------


Expand Down
21 changes: 12 additions & 9 deletions tests/e2e/core/dashboard/test_api_session_chat_limits.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,15 +22,18 @@
import pytest

from tests.e2e.core.dashboard._helpers import Sandbox, make_sandbox
from tests.e2e.core.dashboard._issue_helpers import GatewayApiServer, Issue120937, xfail_known
from tests.e2e.core._pending_fixes import known_gate
from tests.e2e.core.dashboard._issue_helpers import GatewayApiServer, Issue120937

pytestmark = pytest.mark.skipif(not sys.platform.startswith("linux"), reason="reaper reads /proc")
TURN_TIMEOUT = 90.0

KNOWN: dict[str, tuple[str, type]] = {
# test name -> (the bug's own failure signature, "#issue reason"); see _pending_fixes.known_failure.
KNOWN: dict[str, tuple[str, str]] = {
"test_session_chat_100k_message_reaches_model_whole_or_is_rejected": (
r"^#120937 /api/sessions/\{id\}/chat answered 200 but the model received 65536 of 100000 chars",
"#120937 POST /api/sessions/{id}/chat silently truncates a string message to 65,536 chars "
"(200, no error, no flag)", Issue120937),
"(200, no error, no flag)"),
}


Expand Down Expand Up @@ -128,7 +131,6 @@ def test_session_chat_60k_and_runs_100k_reach_model_whole(gateway: tuple[Sandbox
assert run.get("status") == "completed", f"/v1/runs/{run_id} -> {run}"


@xfail_known(KNOWN, "test_session_chat_100k_message_reaches_model_whole_or_is_rejected")
def test_session_chat_100k_message_reaches_model_whole_or_is_rejected(
gateway: tuple[Sandbox, GatewayApiServer]) -> None:
sb, gw = gateway
Expand All @@ -140,8 +142,9 @@ def test_session_chat_100k_message_reaches_model_whole_or_is_rejected(
return
assert status == 200, f"/chat 100k -> {status}: {str(body)[:500]}\n{gw.log_tail()}"
seen = _model_saw(sb, head)
if len(seen) < len(msg) or tail not in seen:
raise Issue120937(
f"#120937 /api/sessions/{{id}}/chat answered 200 but the model received {len(seen)} of "
f"{len(msg)} chars (tail canary {'present' if tail in seen else 'MISSING'}; stored user "
f"message length {_stored_user_len(gw, sid, head)}; response keys {sorted(body)[:12]})")
with known_gate(KNOWN, "test_session_chat_100k_message_reaches_model_whole_or_is_rejected", raises=Issue120937):
if len(seen) < len(msg) or tail not in seen:
raise Issue120937(
f"#120937 /api/sessions/{{id}}/chat answered 200 but the model received {len(seen)} of "
f"{len(msg)} chars (tail canary {'present' if tail in seen else 'MISSING'}; stored user "
f"message length {_stored_user_len(gw, sid, head)}; response keys {sorted(body)[:12]})")
21 changes: 14 additions & 7 deletions tests/e2e/core/dashboard/test_dashboard_mcp_install.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,19 +28,24 @@
import yaml

from tests.e2e.core.dashboard._helpers import Sandbox, make_sandbox
from tests.e2e.core.dashboard._issue_helpers import Issue120527, PtyDashboard, xfail_known
from tests.e2e.core._pending_fixes import known_gate
from tests.e2e.core.dashboard._issue_helpers import Issue120527, PtyDashboard

pytestmark = pytest.mark.skipif(not sys.platform.startswith("linux"), reason="pty stdin + /proc reaper")
INSTALL_DEADLINE_S = 30.0 # a healthy install (write config + spawn/probe a local server) takes ~2 s
FOLLOWUP_DEADLINE_S = 5.0
FIXTURE = Path(__file__).with_name("fixture_catalog_mcp.py")

# Open issues this file encodes. A strict xfail XPASSes (and fails) the moment the fix lands, so the
# entry is removed with the fix; ``raises`` excuses only the issue's own assertion class.
KNOWN: dict[str, tuple[str, type]] = {
# Open issues this file encodes: test name -> (the bug's own failure signature, "#issue reason"). The
# gate (_pending_fixes.known_failure) excuses only Issue120527 with that message and passes once the
# fix lands; delete the entry then.
KNOWN: dict[str, tuple[str, str]] = {
"test_catalog_install_from_tty_launched_dashboard_never_wedges_the_server": (
# A slow install whose follow-ups both answer is not the wedge; it must stay red.
r"^dashboard wedged by an MCP catalog install \(stdin=/dev/pts/\d+\): install -> ReadTimeout after [^;]*; "
r"(?!GET /api/config -> HTTP 200; /api/ws -> accepted\n)",
"#120527 dashboard MCP catalog install reaches the interactive tool checklist on the serve "
"worker thread when stdin is a TTY and wedges _SKILLS_PROFILE_LOCK", Issue120527),
"worker thread when stdin is a TTY and wedges _SKILLS_PROFILE_LOCK"),
}


Expand Down Expand Up @@ -144,6 +149,8 @@ def test_catalog_install_from_headless_dashboard_completes_with_all_probed_tools
assert block.get("enabled") is True and "tools" not in block, f"unexpected tool filter: {block}"


@xfail_known(KNOWN, "test_catalog_install_from_tty_launched_dashboard_never_wedges_the_server")
def test_catalog_install_from_tty_launched_dashboard_never_wedges_the_server(sb: Sandbox, tmp_path: Path) -> None:
_install_and_probe(sb, tmp_path, tty=True, wedge_exc=Issue120527)
# Issue120527 is raised only at the wedge verdict, after the install and both follow-ups settled.
with known_gate(KNOWN, "test_catalog_install_from_tty_launched_dashboard_never_wedges_the_server",
raises=Issue120527):
_install_and_probe(sb, tmp_path, tty=True, wedge_exc=Issue120527)
52 changes: 5 additions & 47 deletions tests/e2e/core/delivery/_pending_fixes.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,14 +3,13 @@
A plain ``xfail(strict=True)`` turns main red the moment its fix merges (XPASS), and a
non-strict one guards nothing. Instead, each gap here has a PROBE: a few lines that reproduce
the defect's mechanism on the tree under test, in a throwaway interpreter with its own
``HOME``/``HERMES_HOME`` (no state leaks into the suite process). ``expect_gap`` applies a
strict xfail only while the probe still reproduces the defect; once the fix is in the tree the
cell runs as a plain test, so it must pass. Whichever lands first, suite or fix, main stays
green, and a probe that disagrees with the end-to-end cell still fails loudly (XPASS, or a
real failure) instead of hiding.
``HOME``/``HERMES_HOME`` (no state leaks into the suite process). A cell asks ``gap_open`` and
tolerates the gap only while the probe still reproduces the defect; once the fix is in the tree
the cell runs as a plain test, so it must pass. Whichever lands first, suite or fix, main stays
green, and a probe that disagrees with the end-to-end cell still fails loudly instead of hiding.

Probes exercise behaviour only (never read source text). When a fix has landed, delete its
entry and every ``expect_gap`` / ``gap_open`` call naming it.
entry and every ``gap_open`` call naming it.

A gap with no fix PR yet has no probe: ``known_failure`` is a run-time xfail keyed on the gap's
own assertion message, so the cell XFAILs only while it fails exactly that way, fails loudly on
Expand Down Expand Up @@ -48,40 +47,6 @@
jobs._hermes_now = lambda: datetime.fromisoformat("2026-03-07T09:00:30-05:00")
nxt = jobs.compute_next_run({"kind": "cron", "expr": "0 9 * * *"})
print("fixed" if nxt == "2026-03-08T09:00:00-04:00" else "open")
'''),
# A failed first stream send disabled edits but left no message id, so the next tick sent a
# second first send: an uneditable partial preview stayed visible next to the final reply.
120315: ({}, r'''
import asyncio
from types import SimpleNamespace
from unittest.mock import AsyncMock, MagicMock
from gateway.stream_consumer import GatewayStreamConsumer, StreamConsumerConfig

async def main():
delivered, results = [], iter([SimpleNamespace(success=False, error="timeout")])
async def send(**kw):
r = next(results, None) or SimpleNamespace(success=True, message_id=f"m{len(delivered)}")
if r.success:
delivered.append(kw["content"])
return r
adapter = MagicMock()
adapter.send = AsyncMock(side_effect=send)
adapter.edit_message = AsyncMock(return_value=SimpleNamespace(success=True))
adapter.MAX_MESSAGE_LENGTH = 4096
c = GatewayStreamConsumer(adapter, "chat", StreamConsumerConfig(edit_interval=0.01,
buffer_threshold=5, cursor=""))
c.on_delta("preview never landed ")
task = asyncio.create_task(c.run())
await asyncio.sleep(0.08)
c.on_delta("and more streamed text ")
await asyncio.sleep(0.08)
c.on_delta("then the end.")
c.finish()
await asyncio.wait_for(task, timeout=10)
return delivered

print("fixed" if asyncio.run(main()) == ["preview never landed and more streamed text then the end."]
else "open")
'''),
}

Expand All @@ -106,13 +71,6 @@ def gap_open(pr: int) -> bool:
return verdict == ["open"]


def expect_gap(request, pr: int, reason: str) -> None:
"""Strict xfail for this cell while #``pr``'s defect reproduces; a plain test once it doesn't."""
assert f"#{pr}" in reason, f"reason for a #{pr} gap must name the PR: {reason!r}"
if gap_open(pr):
request.applymarker(pytest.mark.xfail(strict=True, reason=reason))


@contextlib.contextmanager
def known_failure(pattern: str, reason: str,
on_xfail: Optional[Callable[[], None]] = None) -> Iterator[None]:
Expand Down
18 changes: 2 additions & 16 deletions tests/e2e/core/delivery/test_messaging_exactly_once.py
Original file line number Diff line number Diff line change
Expand Up @@ -46,7 +46,7 @@
visible_copies,
wait_until,
)
from tests.e2e.core.delivery._pending_fixes import expect_gap, known_failure
from tests.e2e.core.delivery._pending_fixes import known_failure
from tests.fakes.fake_llm_provider import FakeLLMServer, StallMidStream, Text, ToolCall

pytestmark = pytest.mark.skipif(sys.platform == "win32", reason="SIGKILL + process-group restart harness")
Expand Down Expand Up @@ -252,18 +252,6 @@ def _say(body: str, **kw):
}


# Streaming gaps #120315 fixes: an overflowing first send keeps the " (n/n)" indicator on the live
# tail that later chunks extend / a sealed remainder over the limit is re-sent; a failed first send
# leaves an uneditable partial preview next to the final. Every cell here fails every run while the
# gap is open (the chunk pacing above puts each chunk in its own consumer tick).
STREAM_OVERFLOW_GAP = "LIVE GAP (fixed by #120315): streamed overflow / failed first send"
STREAM_OVERFLOW_GAP_CELLS = {
("burst_overflow", "fk_tg"), ("burst_overflow", "fk_dc"),
("long_streamed", "fk_dc"),
("stream_timeout_first_send", "fk_tg"), ("stream_timeout_first_send", "fk_dc"),
}


STREAM_ACK_LOST_GAP = (
"LIVE GAP (#53449/#25349 family): on a streaming platform the stream consumer's first send is the "
"whole short reply; when the platform accepts it but the ack is lost (timeout), the consumer reports "
Expand All @@ -284,7 +272,7 @@ def faults_fired(gw: GatewayProcess, aid: str) -> List[dict]:

@pytest.mark.parametrize("platform", list(PLATFORMS))
@pytest.mark.parametrize("fault", list(FAULTS))
def test_delivery_fault_matrix(gw, director, platform, fault, request):
def test_delivery_fault_matrix(gw, director, platform, fault):
kind, op, target, build, expect = FAULTS[fault]
token = f"{platform}.{fault}"
chat = f"m-{platform}-{fault}"
Expand All @@ -303,8 +291,6 @@ def test_delivery_fault_matrix(gw, director, platform, fault, request):
f"{token} complete reply\n" + dump(gw, platform, chat), proc=gw.proc, log=gw.log)
if kind is not None:
assert faults_fired(gw, aid), f"injected {kind} never hit a platform call\n{dump(gw, platform, chat)}"
if (fault, platform) in STREAM_OVERFLOW_GAP_CELLS:
expect_gap(request, 120315, STREAM_OVERFLOW_GAP)
guard = (known_failure(STREAM_ACK_LOST_SIGNATURE, STREAM_ACK_LOST_GAP,
on_xfail=lambda: XFAILED_TOKENS.add(token))
if fault == "ack_lost" and STREAMING[platform] else contextlib.nullcontext())
Expand Down
18 changes: 6 additions & 12 deletions tests/e2e/core/history/test_prefix_stability.py
Original file line number Diff line number Diff line change
Expand Up @@ -78,12 +78,10 @@
],
}

KNOWN_BROKEN: dict[str, str] = {}


class ToolsArrayDrift(Exception):
"""The tools array changed between requests of one session. Not an AssertionError: the
known-bug xfail matches only this, so every other invariant still fails the test."""
"""The tools array changed between requests of one session. Its own type, so a future tools-drift
bug can be gated with ``known_failure(..., raises=ToolsArrayDrift)`` without excusing any other
invariant."""


@pytest.fixture
Expand Down Expand Up @@ -166,11 +164,7 @@ def run_journey(world: dict, hops: list[Hop]) -> tuple[str, list[tuple[int, str]
return sid, openings, compaction_idx


@pytest.mark.parametrize("journey", [
pytest.param(name, marks=pytest.mark.xfail(strict=True, raises=ToolsArrayDrift, reason=KNOWN_BROKEN[name]))
if name in KNOWN_BROKEN else name
for name in JOURNEYS
])
@pytest.mark.parametrize("journey", list(JOURNEYS))
def test_request_prefix_is_byte_stable_across_processes(world, journey):
sid, openings, compaction_idx = run_journey(world, JOURNEYS[journey])
srv, home = world["srv"], world["hermes_home"]
Expand All @@ -180,8 +174,8 @@ def test_request_prefix_is_byte_stable_across_processes(world, journey):
def where(i: int) -> str:
return f"request {i} (in {max((o for o in openings if o[0] <= i), default=(0, '?'))[1]})"

# System prompt + messages first; the tools array is checked last, on its own, so a known
# tools-drift xfail cannot mask a message-prefix, usage or integrity regression.
# System prompt + messages first; the tools array is checked last, on its own, so a gated
# tools-drift bug cannot mask a message-prefix, usage or integrity regression.
breaks = prefix_breaks(main, tools=False)
unexpected = [(i, why) for i, why in breaks if i != compaction_idx]
assert not unexpected, "prompt-cache prefix broke outside the compaction boundary:\n" + "\n".join(
Expand Down
Loading
Loading