diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 30a6dae01a6f0..427b99deae425 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -60,6 +60,8 @@ from agent.runtime_cwd import resolve_agent_cwd from agent.message_sanitization import ( close_interrupted_tool_sequence, + is_interrupt_close_row, + provider_owns_transcript, _repair_tool_call_arguments, coalesce_tool_call_id, _sanitize_messages_non_ascii, @@ -2837,11 +2839,19 @@ def run_conversation( ) api_messages = [] + # t_f40dc54a: a provider that keeps its own transcript (bridge relay + # over a resident CLI session) must not be sent the harness-authored + # interrupt-close row — it is a reply the provider never produced, and + # the relay's coherence gate answers it with a full-history re-mint. + # Looked up once per request; fail-open (row sent) on any error. + _omit_interrupt_close = provider_owns_transcript(agent.provider) for idx, msg in enumerate(messages): # Metadata-only provider events are durable UI rows, never system # instructions in the provider request. if is_metadata_only_tool_notice(msg): continue + if _omit_interrupt_close and is_interrupt_close_row(msg): + continue # Structural clone, NOT msg.copy(): every in-place transform # below (canonicalize/repair, surrogate + non-ASCII sanitizers, diff --git a/agent/message_sanitization.py b/agent/message_sanitization.py index 397fa8a78ed23..cf66e909b1d9d 100644 --- a/agent/message_sanitization.py +++ b/agent/message_sanitization.py @@ -433,6 +433,37 @@ def close_interrupted_tool_sequence( return True +def is_interrupt_close_row(msg: Any) -> bool: + """True for a harness-authored interrupt-close assistant row.""" + return ( + isinstance(msg, dict) + and msg.get("role") == "assistant" + and ( + msg.get("_interrupt_close") is True + or msg.get("finish_reason") == _INTERRUPT_CLOSE_FINISH_REASON + ) + ) + + +def provider_owns_transcript(provider: Any) -> bool: + """Whether the provider profile for ``provider`` opts out of receiving + interrupt-close rows (``ProviderProfile.owns_transcript``). + + A relay over a resident CLI session keeps its own transcript and refuses + to resume when the client history carries an assistant reply it never + produced; the interrupt-close placeholder is exactly that (one + full-history re-mint per gateway restart per session, 2026-09-25). + Fail-open: any lookup problem means the row is sent as before. + """ + try: + from providers import get_provider_profile + + profile = get_provider_profile(str(provider or "")) + except Exception: + return False + return bool(getattr(profile, "owns_transcript", False)) + + def _strip_non_ascii(text: str) -> str: """Remove non-ASCII characters, replacing with closest ASCII equivalent or removing. diff --git a/providers/base.py b/providers/base.py index 2df608f0e3690..19bf345e7fa5b 100644 --- a/providers/base.py +++ b/providers/base.py @@ -78,6 +78,20 @@ class ProviderProfile: # top-level fields rather than ignoring them. supports_prompt_cache_key: bool = False + # owns_transcript: the provider keeps its OWN copy of the conversation + # (a relay over a resident CLI session, resumed by a routing key) and + # reconciles what the harness sends against it. For such a lane the + # harness-authored interrupt-close row (``_interrupt_close``: the + # "Operation interrupted." placeholder or a partial reply appended by + # ``close_interrupted_tool_sequence`` on a /stop or gateway restart) is a + # reply the provider never produced — its coherence gate refuses to + # resume and re-sends the whole history into a fresh session (one full + # prompt-cache rewrite per restart per session, 2026-09-25). Opt-in: + # profiles that set this have the row OMITTED from the wire; persisted + # history is untouched and strict-alternation providers (which need the + # row, #48879) keep the default. + owns_transcript: bool = False + # ── Model catalog ───────────────────────────────────────── # fallback_models: curated list shown in /model picker when live fetch fails. # Only agentic models that support tool calling should appear here. diff --git a/tests/agent/test_interrupt_close_owns_transcript.py b/tests/agent/test_interrupt_close_owns_transcript.py new file mode 100644 index 0000000000000..7e67282a459b9 --- /dev/null +++ b/tests/agent/test_interrupt_close_owns_transcript.py @@ -0,0 +1,105 @@ +"""t_f40dc54a: providers that own their transcript never receive the +harness-authored interrupt-close row; every other provider still does. + +Live incident (2026-09-25, sub-vps-1, bpx): after a gateway restart the +harness closed the interrupted turn with an assistant row the resident CLI +session never wrote. The bridge's continuation gate saw an assistant reply +after the last tool call that was not on disk ("final-reply-produced-elsewhere") +and re-sent the whole 828k history into a fresh session -- one full +prompt-cache rewrite per restart per session. +""" +from dataclasses import replace + +import pytest + +from agent.message_sanitization import ( + close_interrupted_tool_sequence, + is_interrupt_close_row, + provider_owns_transcript, +) +from providers.base import ProviderProfile + + +def _history(): + msgs = [ + {"role": "user", "content": "run the migration"}, + { + "role": "assistant", + "content": "", + "tool_calls": [{"id": "call-1", "type": "function", + "function": {"name": "shell", "arguments": "{}"}}], + }, + {"role": "tool", "tool_call_id": "call-1", "content": "halfway"}, + ] + assert close_interrupted_tool_sequence(msgs) is True + msgs.append({"role": "user", "content": "status?"}) + return msgs + + +def test_row_detection(): + msgs = _history() + assert is_interrupt_close_row(msgs[3]) + assert not is_interrupt_close_row({"role": "assistant", "content": "Operation interrupted."}) + assert not is_interrupt_close_row({"role": "user", "finish_reason": "interrupt_close"}) + assert is_interrupt_close_row({"role": "assistant", "content": "x", "finish_reason": "interrupt_close"}) + + +def test_default_profile_does_not_own_transcript(): + assert ProviderProfile(name="x").owns_transcript is False + + +@pytest.fixture +def registry(monkeypatch): + import providers + + profiles = { + "bridge-lane": ProviderProfile(name="bridge-lane", owns_transcript=True), + "native-lane": ProviderProfile(name="native-lane", api_mode="anthropic_messages"), + } + monkeypatch.setattr(providers, "get_provider_profile", lambda n: profiles.get(n)) + return profiles + + +def test_provider_owns_transcript_lookup(registry): + assert provider_owns_transcript("bridge-lane") is True + assert provider_owns_transcript("native-lane") is False + assert provider_owns_transcript("unknown") is False + assert provider_owns_transcript(None) is False + + +def test_lookup_fails_open(monkeypatch): + import providers + + def boom(_): + raise RuntimeError("registry down") + + monkeypatch.setattr(providers, "get_provider_profile", boom) + assert provider_owns_transcript("bridge-lane") is False + + +def _wire(provider, msgs): + """Mirror of the conversation_loop filter (same two helpers).""" + omit = provider_owns_transcript(provider) + return [m for m in msgs if not (omit and is_interrupt_close_row(m))] + + +def test_bridge_lane_omits_close_row_native_keeps_it(registry): + msgs = _history() + bridge = _wire("bridge-lane", msgs) + native = _wire("native-lane", msgs) + assert [m["role"] for m in bridge] == ["user", "assistant", "tool", "user"] + assert not any(is_interrupt_close_row(m) for m in bridge) + assert [m["role"] for m in native] == ["user", "assistant", "tool", "assistant", "user"] + # persisted history untouched + assert is_interrupt_close_row(msgs[3]) + + +def test_conversation_loop_wires_the_filter(): + """Contract: the send-path loop consults both helpers before cloning rows.""" + import inspect + + from agent import conversation_loop + + src = inspect.getsource(conversation_loop) + assert "provider_owns_transcript(agent.provider)" in src + assert "is_interrupt_close_row(msg)" in src