From 96aa25eb903ecbc9fdf90cd4ed5e16549b0a2ce1 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Thu, 27 Aug 2026 17:44:04 +0000 Subject: [PATCH 01/16] add context compaction to the bash and rlm harnesses Optional CompactionConfig on both in-house harnesses: compact into a handoff summary at summarize_at_tokens, or at 90% of the model context window when the provider advertises one. The threshold is discovered by the agent loops themselves (the bash program reads the provider's /models card; nano-rlm's engine already does), so the interception server gains a stateless GET /v1/models relay serving every dialect. On a provider overflow error the loops compact and retry once, learning the threshold from the error message; an oversized checkpoint request drops the newest tool results one at a time until it fits. RLM's summarize_at_tokens moves into the compaction config and crosses ACP flat; nano-rlm pinned at f5c14aa. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/harness.py | 19 +++ verifiers/v1/harnesses/bash/program.py | 204 ++++++++++++++++++++++++- verifiers/v1/harnesses/rlm/harness.py | 49 +++--- verifiers/v1/interception/server.py | 38 ++++- 4 files changed, 271 insertions(+), 39 deletions(-) diff --git a/verifiers/v1/harnesses/bash/harness.py b/verifiers/v1/harnesses/bash/harness.py index 92273025cc..4100dcec47 100644 --- a/verifiers/v1/harnesses/bash/harness.py +++ b/verifiers/v1/harnesses/bash/harness.py @@ -2,6 +2,9 @@ import os from pathlib import Path +from pydantic import PositiveInt +from pydantic_config import BaseConfig + from verifiers.v1.clients import ModelContext from verifiers.v1.configs.harness import HarnessConfig from verifiers.v1.dialects.chat import message_to_wire @@ -27,6 +30,14 @@ ) +class CompactionConfig(BaseConfig): + """Context compaction policy for the bash agent loop.""" + + summarize_at_tokens: PositiveInt | None = None + """Compact at this token count. When unset, use 90% of the model context window when + the provider advertises it.""" + + class BashHarnessConfig(HarnessConfig): edit: bool = True """Offer the local `edit` tool (single-occurrence string replacement in a file) alongside @@ -37,6 +48,9 @@ class BashHarnessConfig(HarnessConfig): eval environment; the key is handed to the program over argv (like the interception secret) so the agent's `bash` subprocesses don't inherit it.""" + compaction: CompactionConfig | None = None + """Context compaction policy. Set an empty config to use automatic thresholds.""" + class BashHarness(Harness[BashHarnessConfig]): APPENDS_SYSTEM_PROMPT = True @@ -77,6 +91,11 @@ async def launch( ] if tool_interception_url: args.append(f"--tool-interception-url={tool_interception_url}") + if self.config.compaction is not None: + args.append("--compaction") + threshold = self.config.compaction.summarize_at_tokens + if threshold is not None: + args.append(f"--summarize-at-tokens={threshold}") if self.config.edit: args.append("--edit") if self.config.search: diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index 7a31c6cad5..c53a375cf0 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -7,12 +7,14 @@ import argparse import asyncio import json +import re import subprocess from contextlib import AsyncExitStack, asynccontextmanager, suppress from pathlib import Path +from typing import Any import httpx -from openai import AsyncOpenAI +from openai import APIError, AsyncOpenAI, BadRequestError from tenacity import AsyncRetrying, stop_after_attempt, wait_exponential_jitter SERPER_URL = "https://google.serper.dev/search" @@ -20,6 +22,63 @@ MCP_CALL_ATTEMPTS = 6 MCP_TIMEOUT = 600.0 +CHECKPOINT_COMPACTION_PROMPT = """You are performing a CONTEXT CHECKPOINT COMPACTION. Create a handoff summary for another LLM that will resume the task. + +Include: +- Current progress and key decisions made +- Important context, constraints, or user preferences +- What remains to be done (clear next steps) +- Any critical data, examples, or references needed to continue + +Be concise, structured, and focused on helping the next LLM seamlessly continue the work. + +Reply with the summary as plain text. Do not call any tools - summarize from the conversation as it stands.""" + +POST_COMPACTION_FRAMING = """Another language model started to solve this problem and produced \ +a summary of its thinking process. Use this to build on the work \ +that has already been done and avoid duplicating work. Here is \ +the summary produced by the other language model, use the \ +information in this summary to assist with your own analysis:""" + +COMPACTED_TOOL_RESULT = "[tool output dropped because it exceeded the context limit]" + +CONTEXT_WINDOW_PATTERNS = ( + re.compile( + r"(?:maximum|max(?:imum)?)[^.\n]{0,40}(?:context length|context window)" + r"[^\d]{0,20}([\d,]+)", + re.IGNORECASE, + ), + re.compile( + r"[\"']?(?:max_model_len|context_length)[\"']?\s*[:=]\s*([\d,]+)", + re.IGNORECASE, + ), +) + +CONTEXT_WINDOW_FIELDS = ( + "max_model_len", + "context_length", + "context_window", + "max_context_length", +) + + +async def discover_threshold(client: AsyncOpenAI, model: str) -> int | None: + """90% of the model context window, when the provider's model card advertises one.""" + try: + # The SDK needs a parameterized mapping type to parse into (bare `dict` fails). + payload = await client.get("/models", cast_to=dict[str, Any]) + except APIError: + return None + for card in payload.get("data") or []: + if not isinstance(card, dict) or card.get("id") != model: + continue + for field in CONTEXT_WINDOW_FIELDS: + value = card.get(field) + if isinstance(value, int) and not isinstance(value, bool) and value > 0: + return max(1, value * 9 // 10) + break + return None + BASH_TOOL = { "type": "function", @@ -178,12 +237,122 @@ def run_edit(path: str, old_str: str, new_str: str) -> str: async def chat( - client: AsyncOpenAI, model: str, messages: list[dict], tools: list[dict] + client: AsyncOpenAI, + model: str, + messages: list[dict], + tools: list[dict], + *, + tool_choice: str | None = None, ): - completion = await client.chat.completions.create( - model=model, messages=messages, tools=tools or None + kwargs = {"model": model, "messages": messages, "tools": tools or None} + if tools and tool_choice is not None: + kwargs["tool_choice"] = tool_choice + return await client.chat.completions.create(**kwargs) + + +def context_error(error: BadRequestError) -> tuple[bool, int | None]: + details = f"{error} {error.body or ''}" + overflow = any( + marker in details.casefold() + for marker in ( + "request entity too large", + "context_length", + "context length", + "context window", + "prompt is too long", + "too many tokens", + "token limit exceeded", + ) ) - return completion.choices[0].message + for pattern in CONTEXT_WINDOW_PATTERNS: + match = pattern.search(details) + if match: + context_window = int(match.group(1).replace(",", "")) + return overflow, max(1, context_window * 9 // 10) + return overflow, None + + +def drop_latest_tool_result(messages: list[dict]) -> bool: + """Replace one tool result so the checkpoint request can fit in context.""" + for index in range(len(messages) - 1, -1, -1): + message = messages[index] + if message.get("role") != "tool": + continue + if message.get("content") == COMPACTED_TOOL_RESULT: + continue + messages[index] = {**message, "content": COMPACTED_TOOL_RESULT} + return True + return False + + +def estimated_tokens(chars: str) -> int: + """Rough token count at four characters per token.""" + return (len(chars) + 3) // 4 + + +def context_tokens(completion) -> int: + usage = completion.usage + if usage is None: + return 0 + return (usage.prompt_tokens or 0) + (usage.completion_tokens or 0) + + +class Compactor: + """Compact once and retry once when a model turn exhausts its context.""" + + def __init__(self, client, model, tools, enabled, threshold): + self.client = client + self.model = model + self.tools = tools + self.enabled = enabled + self.threshold = threshold + + def reached(self, completion, extra_tokens: int = 0) -> bool: + return ( + self.enabled + and self.threshold is not None + and context_tokens(completion) + extra_tokens >= self.threshold + ) + + async def complete(self, messages: list[dict]): + try: + completion = await chat(self.client, self.model, messages, self.tools) + except BadRequestError as error: + overflow, threshold = context_error(error) + if not self.enabled or not overflow: + raise + if self.threshold is None: + self.threshold = threshold + else: + choice = completion.choices[0] + if choice.finish_reason != "length" or not self.reached(completion): + return completion, messages + + messages = await self.compact(messages) + completion = await chat(self.client, self.model, messages, self.tools) + return completion, messages + + async def compact(self, messages: list[dict]) -> list[dict]: + system = [message for message in messages if message.get("role") == "system"] + while True: + checkpoint = [ + *messages, + {"role": "user", "content": CHECKPOINT_COMPACTION_PROMPT}, + ] + try: + completion = await chat( + self.client, + self.model, + checkpoint, + self.tools, + tool_choice="none", + ) + summary = completion.choices[0].message.content or "" + framed = POST_COMPACTION_FRAMING + "\n\n" + summary + return [*system, {"role": "user", "content": framed}] + except BadRequestError as error: + if not context_error(error)[0] or not drop_latest_tool_result(messages): + raise @asynccontextmanager @@ -328,6 +497,8 @@ def parse_args() -> argparse.Namespace: parser.add_argument("--initial-messages-file", default="") parser.add_argument("--mcp-config", default="") parser.add_argument("--tool-interception-url", default="") + parser.add_argument("--compaction", action="store_true") + parser.add_argument("--summarize-at-tokens", type=int) parser.add_argument("--edit", action="store_true") parser.add_argument("--search", action="store_true") parser.add_argument("--serper-key", default="") @@ -372,11 +543,23 @@ async def main() -> None: messages.extend(initial) elif args.prompt: messages.append({"role": "user", "content": args.prompt}) + compactor = Compactor( + client, + args.model, + tools, + args.compaction, + args.summarize_at_tokens, + ) + if compactor.enabled and compactor.threshold is None: + compactor.threshold = await discover_threshold(client, args.model) while True: - message = await chat(client, args.model, messages, tools) + completion, messages = await compactor.complete(messages) + choice = completion.choices[0] + message = choice.message messages.append(message.model_dump(exclude_none=True)) if not message.tool_calls: break + tool_result_tokens = 0 for call in message.tool_calls: name = call.function.name tool_message = { @@ -395,7 +578,11 @@ async def main() -> None: tool_message, ) if decision["action"] == "rewrite": - messages.append(decision["message"]) + rewritten = decision["message"] + messages.append(rewritten) + tool_result_tokens += estimated_tokens( + str(rewritten.get("content", "")) + ) continue try: tool_args = json.loads(call.function.arguments or "{}") @@ -440,6 +627,9 @@ async def main() -> None: if decision["action"] == "rewrite": tool_message = decision["message"] messages.append(tool_message) + tool_result_tokens += estimated_tokens(str(tool_message["content"])) + if compactor.reached(completion, tool_result_tokens): + messages = await compactor.compact(messages) if tool_client is not None: await tool_client.aclose() diff --git a/verifiers/v1/harnesses/rlm/harness.py b/verifiers/v1/harnesses/rlm/harness.py index 639d556c2e..a503828542 100644 --- a/verifiers/v1/harnesses/rlm/harness.py +++ b/verifiers/v1/harnesses/rlm/harness.py @@ -1,11 +1,11 @@ """RLM over ACP, with MCP tools exposed as pre-imported IPython skills.""" import logging -import random import shlex from typing import Literal from pydantic import BaseModel, ConfigDict, Field, PositiveInt, model_validator +from pydantic_config import BaseConfig from verifiers.v1.acp import ACPConfig, ACPHarness, ACPTurn, JsonObject from verifiers.v1.clients import ModelContext @@ -34,30 +34,24 @@ class _SessionSnapshot(BaseModel): metrics: dict[str, int | float] +class CompactionConfig(BaseConfig): + """Context compaction policy for the RLM agent loop.""" + + summarize_at_tokens: PositiveInt | None = None + """Compact at this token count. When unset, use 90% of the model context window when + the provider advertises it.""" + + class RLMHarnessConfig(HarnessConfig): - version: str = Field( - default="d4ce3e10e63b359f4f3d432d58a77471e9e21fe7", min_length=1 - ) + version: str = Field(default="f5c14aa", min_length=1) """Git ref (branch, tag, or commit) of nano-rlm to install.""" max_depth: int = 0 """Recursion depth RLM may spawn sub-harnesses to.""" builtin_skills: list[BuiltinSkill] = Field(default_factory=list) """Built-in rlm skills to enable (RLM_SKILLS), e.g. `["edit"]`; empty enables none. The tool set is fixed (ipython); the base `skills` field takes SKILL.md paths.""" - summarize_at_tokens: PositiveInt | tuple[PositiveInt, PositiveInt] | None = None - """Auto-compaction threshold (RLM_SUMMARIZE_AT_TOKENS): compact the context once it grows - past this many tokens. An int is a fixed threshold; a `(lo, hi)` pair draws a per-group - threshold (seeded by the task index, so a task's rollouts share one draw and tasks vary). - `None` disables auto-compaction; ints must be positive.""" - - @model_validator(mode="after") - def validate_range(self) -> "RLMHarnessConfig": - value = self.summarize_at_tokens - if isinstance(value, tuple) and value[0] > value[1]: - raise ValueError( - "`summarize_at_tokens` range must be (lo, hi) with lo <= hi." - ) - return self + compaction: CompactionConfig | None = None + """Context compaction policy. Set an empty config to use automatic thresholds.""" @model_validator(mode="after") def reject_disabled_tools(self) -> "RLMHarnessConfig": @@ -97,16 +91,6 @@ async def setup(self, runtime: Runtime) -> None: raise RuntimeError(f"rlm install failed: {result.stderr.strip()[-500:]}") await super().setup(runtime) - def summarize_threshold(self, task_idx: int | None) -> int | None: - """Resolve a fixed or per-task compaction threshold.""" - value = self.config.summarize_at_tokens - if value is None: - return None - if isinstance(value, tuple): - lo, hi = value - return random.Random(task_idx or 0).randint(lo, hi) - return value - def _runtime_metadata( self, ctx: ModelContext, @@ -114,9 +98,9 @@ def _runtime_metadata( runtime: Runtime, endpoint: str, secret: str, - data: TaskData, system_prompt: str | None, ) -> JsonObject: + compaction = self.config.compaction payload = { "session_id": trace.id, "model": ctx.model, @@ -126,7 +110,10 @@ def _runtime_metadata( }, "policy": { "max_depth": self.config.max_depth, - "summarize_at_tokens": self.summarize_threshold(data.idx), + "compaction": compaction is not None, + "summarize_at_tokens": ( + compaction.summarize_at_tokens if compaction else None + ), "max_concurrent_subagents": max(4, self.config.max_depth), }, "system_prompt_path": None, @@ -153,7 +140,7 @@ async def prepare_acp( command=[RLM_BIN, "--acp"], prompt=prompt, session_meta=self._runtime_metadata( - ctx, trace, runtime, endpoint, secret, data, system_prompt + ctx, trace, runtime, endpoint, secret, system_prompt ), ) diff --git a/verifiers/v1/interception/server.py b/verifiers/v1/interception/server.py index fcb5cc9b41..8731cbdadf 100644 --- a/verifiers/v1/interception/server.py +++ b/verifiers/v1/interception/server.py @@ -33,13 +33,15 @@ from tempfile import SpooledTemporaryFile from typing import Literal +import httpx from aiohttp import web from pydantic import ValidationError from pydantic_core import PydanticSerializationError, from_json, to_json from verifiers.v1 import graph from verifiers.v1.clients import Client, resolve_client -from verifiers.v1.configs.client import BaseClientConfig +from verifiers.v1.clients.base import DEFAULT_TIMEOUT, join_url +from verifiers.v1.configs.client import BaseClientConfig, resolve_api_key from verifiers.v1.dialects import DIALECTS, Dialect from verifiers.v1.dialects.base import ( PROVIDER_CAPABILITY_POLICY_CODE, @@ -326,6 +328,9 @@ async def start(self) -> None: app.router.add_post(route, self._handler_for(dialect)) for aux in dialect.aux_routes: app.router.add_post(aux, self._aux_handler_for(dialect, aux)) + # One models route serves every dialect: OpenAI and Anthropic SDKs both list + # models at `GET /v1/models` (the response schema is the upstream's). + app.router.add_get("/v1/models", self.handle_models) # Tool servers use a state-only capability; the model bearer cannot reach these. app.router.add_get("/state", self.handle_state_get) app.router.add_put("/state", self.handle_state_put) @@ -1053,6 +1058,37 @@ async def handle_aux( return web.json_response(dialect.error_body(str(e)), status=502) return web.json_response(result) + async def handle_models(self, request: web.Request) -> web.Response: + """`GET /v1/models`: relay the upstream model listing so agent loops can read a + provider context-window extension (e.g. vLLM's `max_model_len`). The path is shared + by every dialect; only the auth carrier differs, so the bearer is tried per dialect. + A pure relay from the session's endpoint config — never recorded on the trace, and a + failure never fails the rollout.""" + for dialect in DIALECTS: + session = self.sessions.get(dialect.secret(request.headers)) + if session is not None: + break + else: + return web.json_response({"error": "unauthorized"}, status=401) + session.adopt(asyncio.current_task()) + logger.debug("intercept models: id=%s", session.trace.id) + config = session.ctx.client + headers = dict(config.headers or {}) + headers.update(dialect.auth_headers(resolve_api_key(config))) + try: + async with httpx.AsyncClient(timeout=DEFAULT_TIMEOUT) as client: + upstream = await client.get( + join_url(config.base_url, "/v1/models"), headers=headers + ) + except httpx.HTTPError as e: + logger.warning("models call failed: id=%s %s", session.trace.id, e) + return web.json_response(dialect.error_body(str(e)), status=502) + return web.Response( + body=upstream.content, + status=upstream.status_code, + content_type="application/json", + ) + def _session_for( self, request: web.Request, *, allow_service: bool = False ) -> RolloutSession | None: From 25f12a6f61677d127a0453c5a9de0e20693714ee Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Thu, 27 Aug 2026 18:35:08 +0000 Subject: [PATCH 02/16] bump nano-rlm to the main merge Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/rlm/harness.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/verifiers/v1/harnesses/rlm/harness.py b/verifiers/v1/harnesses/rlm/harness.py index a503828542..fa1eb965f0 100644 --- a/verifiers/v1/harnesses/rlm/harness.py +++ b/verifiers/v1/harnesses/rlm/harness.py @@ -43,7 +43,7 @@ class CompactionConfig(BaseConfig): class RLMHarnessConfig(HarnessConfig): - version: str = Field(default="f5c14aa", min_length=1) + version: str = Field(default="f1c51fb", min_length=1) """Git ref (branch, tag, or commit) of nano-rlm to install.""" max_depth: int = 0 """Recursion depth RLM may spawn sub-harnesses to.""" From 339a785f2b4ae1f514f21a62014bb5a44994f0f5 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Thu, 27 Aug 2026 19:36:27 +0000 Subject: [PATCH 03/16] reserve fixed headroom and truncate tool output Compact when 16k tokens remain below the context window instead of at 90% of it - a fixed reserve keeps constant headroom on any window size (small windows keep at least half). Truncate a tool result over 10KB middle-out before it enters the conversation, with a warning naming the original token count and line count, so one giant output can never leap past the reserve and the model knows what was cut. Matches Codex's output policy; the threshold matches pi's reserve design. Pin nano-rlm 4fd3fa2 with the same changes. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/harness.py | 4 +-- verifiers/v1/harnesses/bash/program.py | 34 +++++++++++++++++++++++--- verifiers/v1/harnesses/rlm/harness.py | 6 ++--- 3 files changed, 35 insertions(+), 9 deletions(-) diff --git a/verifiers/v1/harnesses/bash/harness.py b/verifiers/v1/harnesses/bash/harness.py index 4100dcec47..4377497195 100644 --- a/verifiers/v1/harnesses/bash/harness.py +++ b/verifiers/v1/harnesses/bash/harness.py @@ -34,8 +34,8 @@ class CompactionConfig(BaseConfig): """Context compaction policy for the bash agent loop.""" summarize_at_tokens: PositiveInt | None = None - """Compact at this token count. When unset, use 90% of the model context window when - the provider advertises it.""" + """Compact at this token count. When unset, compact when 16k tokens remain below the + model context window when the provider advertises it.""" class BashHarnessConfig(HarnessConfig): diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index c53a375cf0..4178255cfa 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -22,6 +22,12 @@ MCP_CALL_ATTEMPTS = 6 MCP_TIMEOUT = 600.0 +RESERVE_TOKENS = 16_384 +"""Compact when this many tokens remain below the model context window.""" + +TOOL_OUTPUT_MAX_BYTES = 10_000 +"""Middle-out truncation budget for one tool result before it enters the conversation.""" + CHECKPOINT_COMPACTION_PROMPT = """You are performing a CONTEXT CHECKPOINT COMPACTION. Create a handoff summary for another LLM that will resume the task. Include: @@ -62,8 +68,13 @@ ) +def default_threshold(context_window: int) -> int: + """Leave a fixed reserve below the window; small windows keep at least half.""" + return max(context_window - RESERVE_TOKENS, context_window // 2) + + async def discover_threshold(client: AsyncOpenAI, model: str) -> int | None: - """90% of the model context window, when the provider's model card advertises one.""" + """The compaction threshold, when the provider's model card advertises a context window.""" try: # The SDK needs a parameterized mapping type to parse into (bare `dict` fails). payload = await client.get("/models", cast_to=dict[str, Any]) @@ -75,7 +86,7 @@ async def discover_threshold(client: AsyncOpenAI, model: str) -> int | None: for field in CONTEXT_WINDOW_FIELDS: value = card.get(field) if isinstance(value, int) and not isinstance(value, bool) and value > 0: - return max(1, value * 9 // 10) + return default_threshold(value) break return None @@ -268,7 +279,7 @@ def context_error(error: BadRequestError) -> tuple[bool, int | None]: match = pattern.search(details) if match: context_window = int(match.group(1).replace(",", "")) - return overflow, max(1, context_window * 9 // 10) + return overflow, default_threshold(context_window) return overflow, None @@ -285,6 +296,21 @@ def drop_latest_tool_result(messages: list[dict]) -> bool: return False +def truncate_tool_output(text: str) -> str: + """Keep the head and tail of an oversized tool result and say what was cut.""" + data = text.encode("utf-8") + if len(data) <= TOOL_OUTPUT_MAX_BYTES: + return text + keep = TOOL_OUTPUT_MAX_BYTES // 2 + head = data[:keep].decode("utf-8", errors="ignore") + tail = data[-keep:].decode("utf-8", errors="ignore") + return ( + f"Warning: truncated output (original token count: {estimated_tokens(text)})\n" + f"Total output lines: {text.count(chr(10)) + 1}\n\n" + f"{head}\n[... {len(data) - 2 * keep} bytes truncated ...]\n{tail}" + ) + + def estimated_tokens(chars: str) -> int: """Rough token count at four characters per token.""" return (len(chars) + 3) // 4 @@ -614,7 +640,7 @@ async def main() -> None: ) else: content = f"error: unknown tool {name!r}" - tool_message["content"] = content + tool_message["content"] = truncate_tool_output(content) if args.tool_interception_url: assert tool_client is not None decision = await run_tool_hook( diff --git a/verifiers/v1/harnesses/rlm/harness.py b/verifiers/v1/harnesses/rlm/harness.py index fa1eb965f0..09033ea113 100644 --- a/verifiers/v1/harnesses/rlm/harness.py +++ b/verifiers/v1/harnesses/rlm/harness.py @@ -38,12 +38,12 @@ class CompactionConfig(BaseConfig): """Context compaction policy for the RLM agent loop.""" summarize_at_tokens: PositiveInt | None = None - """Compact at this token count. When unset, use 90% of the model context window when - the provider advertises it.""" + """Compact at this token count. When unset, compact when 16k tokens remain below the + model context window when the provider advertises it.""" class RLMHarnessConfig(HarnessConfig): - version: str = Field(default="f1c51fb", min_length=1) + version: str = Field(default="4fd3fa2", min_length=1) """Git ref (branch, tag, or commit) of nano-rlm to install.""" max_depth: int = 0 """Recursion depth RLM may spawn sub-harnesses to.""" From 514106cb6bdc7ad61f385c7baeacd3798f957376 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Thu, 27 Aug 2026 21:32:53 +0000 Subject: [PATCH 04/16] attribute overflow markers to their providers Each marker now names the API whose error wording it matches, and the unattributable generics are gone - "too many tokens" also matches Bedrock throttling, and bare "context length"/"context window" substrings matched more than they targeted. Pin nano-rlm 3b97900 with the same map. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/program.py | 39 ++++++++++++++++++-------- verifiers/v1/harnesses/rlm/harness.py | 2 +- 2 files changed, 28 insertions(+), 13 deletions(-) diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index 4178255cfa..6c5dd014e1 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -48,6 +48,32 @@ COMPACTED_TOOL_RESULT = "[tool output dropped because it exceeded the context limit]" +# Overflow wording by provider, matched case-insensitively against the full error body. +# Each marker is attributed to the API that produces it; unattributable generics stay out +# (e.g. "too many tokens" also matches Bedrock throttling). +CONTEXT_OVERFLOW_MARKERS = ( + # OpenAI error code "context_length_exceeded"; OpenRouter relays the raw body. + "context_length_exceeded", + # OpenAI Responses/Completions: "Your input exceeds the context window of this model". + "exceeds the context window", + # OpenAI chat: "Input tokens exceed the configured limit of N tokens. Please reduce + # the length of the messages."; Groq words it the same way. + "reduce the length of the messages", + # vLLM: "This model's maximum context length is N tokens"; the renderers pre-flight: + # "Prompt length (N) exceeds maximum context length (M)"; Mistral uses the same words. + "maximum context length", + # Anthropic: "prompt is too long: N tokens > M maximum". + "prompt is too long", + # Anthropic byte-size overflow: HTTP 413 {"type": "request_too_large"}. + "request_too_large", + # HTTP proxies reject an oversized body with 413 "Request Entity Too Large". + "request entity too large", + # Google: "The input token count (N) exceeds the maximum number of tokens allowed (M)". + "exceeds the maximum number of tokens", + # xAI: "This model's maximum prompt length is N but the request contains M tokens". + "maximum prompt length is", +) + CONTEXT_WINDOW_PATTERNS = ( re.compile( r"(?:maximum|max(?:imum)?)[^.\n]{0,40}(?:context length|context window)" @@ -263,18 +289,7 @@ async def chat( def context_error(error: BadRequestError) -> tuple[bool, int | None]: details = f"{error} {error.body or ''}" - overflow = any( - marker in details.casefold() - for marker in ( - "request entity too large", - "context_length", - "context length", - "context window", - "prompt is too long", - "too many tokens", - "token limit exceeded", - ) - ) + overflow = any(marker in details.casefold() for marker in CONTEXT_OVERFLOW_MARKERS) for pattern in CONTEXT_WINDOW_PATTERNS: match = pattern.search(details) if match: diff --git a/verifiers/v1/harnesses/rlm/harness.py b/verifiers/v1/harnesses/rlm/harness.py index 09033ea113..2d91cb457f 100644 --- a/verifiers/v1/harnesses/rlm/harness.py +++ b/verifiers/v1/harnesses/rlm/harness.py @@ -43,7 +43,7 @@ class CompactionConfig(BaseConfig): class RLMHarnessConfig(HarnessConfig): - version: str = Field(default="4fd3fa2", min_length=1) + version: str = Field(default="3b97900", min_length=1) """Git ref (branch, tag, or commit) of nano-rlm to install.""" max_depth: int = 0 """Recursion depth RLM may spawn sub-harnesses to.""" From 5ea11fca02a425ea474f20674559a30a72232681 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Thu, 27 Aug 2026 21:58:15 +0000 Subject: [PATCH 05/16] catch byte-size overflow too The 413 markers could never fire: a 413 arrives as a plain APIStatusError, not BadRequestError. Catch APIStatusError at the compaction sites and gate overflow detection on a deterministic status (400 or 413) so marker-shaped text in a transient failure never triggers a compaction. Pin nano-rlm 4bb5f48 with the same fix. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/program.py | 16 ++++++++-------- verifiers/v1/harnesses/rlm/harness.py | 2 +- 2 files changed, 9 insertions(+), 9 deletions(-) diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index 6c5dd014e1..c9f2d7513e 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -14,7 +14,7 @@ from typing import Any import httpx -from openai import APIError, AsyncOpenAI, BadRequestError +from openai import APIError, APIStatusError, AsyncOpenAI from tenacity import AsyncRetrying, stop_after_attempt, wait_exponential_jitter SERPER_URL = "https://google.serper.dev/search" @@ -48,9 +48,6 @@ COMPACTED_TOOL_RESULT = "[tool output dropped because it exceeded the context limit]" -# Overflow wording by provider, matched case-insensitively against the full error body. -# Each marker is attributed to the API that produces it; unattributable generics stay out -# (e.g. "too many tokens" also matches Bedrock throttling). CONTEXT_OVERFLOW_MARKERS = ( # OpenAI error code "context_length_exceeded"; OpenRouter relays the raw body. "context_length_exceeded", @@ -287,9 +284,12 @@ async def chat( return await client.chat.completions.create(**kwargs) -def context_error(error: BadRequestError) -> tuple[bool, int | None]: +def context_error(error: APIStatusError) -> tuple[bool, int | None]: details = f"{error} {error.body or ''}" - overflow = any(marker in details.casefold() for marker in CONTEXT_OVERFLOW_MARKERS) + # An overflow is deterministic: a 400, or a 413 for a byte-size cap. + overflow = error.status_code in (400, 413) and any( + marker in details.casefold() for marker in CONTEXT_OVERFLOW_MARKERS + ) for pattern in CONTEXT_WINDOW_PATTERNS: match = pattern.search(details) if match: @@ -358,7 +358,7 @@ def reached(self, completion, extra_tokens: int = 0) -> bool: async def complete(self, messages: list[dict]): try: completion = await chat(self.client, self.model, messages, self.tools) - except BadRequestError as error: + except APIStatusError as error: overflow, threshold = context_error(error) if not self.enabled or not overflow: raise @@ -391,7 +391,7 @@ async def compact(self, messages: list[dict]) -> list[dict]: summary = completion.choices[0].message.content or "" framed = POST_COMPACTION_FRAMING + "\n\n" + summary return [*system, {"role": "user", "content": framed}] - except BadRequestError as error: + except APIStatusError as error: if not context_error(error)[0] or not drop_latest_tool_result(messages): raise diff --git a/verifiers/v1/harnesses/rlm/harness.py b/verifiers/v1/harnesses/rlm/harness.py index 2d91cb457f..f7d8950d94 100644 --- a/verifiers/v1/harnesses/rlm/harness.py +++ b/verifiers/v1/harnesses/rlm/harness.py @@ -43,7 +43,7 @@ class CompactionConfig(BaseConfig): class RLMHarnessConfig(HarnessConfig): - version: str = Field(default="3b97900", min_length=1) + version: str = Field(default="4bb5f48", min_length=1) """Git ref (branch, tag, or commit) of nano-rlm to install.""" max_depth: int = 0 """Recursion depth RLM may spawn sub-harnesses to.""" From 7b15d607ba1be5d3d178d143b9c2a23670cecb89 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Fri, 28 Aug 2026 21:17:52 +0000 Subject: [PATCH 06/16] compact from the last good state Final compaction design: proactive at a fixed reserve below a known context window, reactive on attributed 400/413 overflow errors. A rejected checkpoint no longer sheds tool results - it falls back to the last state that passed a threshold check, which by definition holds a full reserve of room; an empty or tool-calling reply is resampled, and after three failed attempts the program ends the run cleanly as a trainable sample instead of crashing. An overflow with no history beyond the task propagates. Tool truncation grows to 20KB and the threshold-learning regexes go away - compaction now requires a known window. Pin nano-rlm b928097 with the same design. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/program.py | 120 ++++++++++++++----------- verifiers/v1/harnesses/rlm/harness.py | 2 +- 2 files changed, 70 insertions(+), 52 deletions(-) diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index c9f2d7513e..b45bec9df4 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -7,7 +7,6 @@ import argparse import asyncio import json -import re import subprocess from contextlib import AsyncExitStack, asynccontextmanager, suppress from pathlib import Path @@ -25,7 +24,11 @@ RESERVE_TOKENS = 16_384 """Compact when this many tokens remain below the model context window.""" -TOOL_OUTPUT_MAX_BYTES = 10_000 +COMPACTION_ATTEMPTS = 3 +"""Checkpoint attempts before compaction fails: a rejected request falls back to the +last good snapshot; an empty or tool-calling reply is resampled.""" + +TOOL_OUTPUT_MAX_BYTES = 20_000 """Middle-out truncation budget for one tool result before it enters the conversation.""" CHECKPOINT_COMPACTION_PROMPT = """You are performing a CONTEXT CHECKPOINT COMPACTION. Create a handoff summary for another LLM that will resume the task. @@ -41,13 +44,12 @@ Reply with the summary as plain text. Do not call any tools - summarize from the conversation as it stands.""" POST_COMPACTION_FRAMING = """Another language model started to solve this problem and produced \ -a summary of its thinking process. Use this to build on the work \ +a summary of its thinking process. You also have access to the state of the tools that \ +were used by that language model. Use this to build on the work \ that has already been done and avoid duplicating work. Here is \ the summary produced by the other language model, use the \ information in this summary to assist with your own analysis:""" -COMPACTED_TOOL_RESULT = "[tool output dropped because it exceeded the context limit]" - CONTEXT_OVERFLOW_MARKERS = ( # OpenAI error code "context_length_exceeded"; OpenRouter relays the raw body. "context_length_exceeded", @@ -71,18 +73,6 @@ "maximum prompt length is", ) -CONTEXT_WINDOW_PATTERNS = ( - re.compile( - r"(?:maximum|max(?:imum)?)[^.\n]{0,40}(?:context length|context window)" - r"[^\d]{0,20}([\d,]+)", - re.IGNORECASE, - ), - re.compile( - r"[\"']?(?:max_model_len|context_length)[\"']?\s*[:=]\s*([\d,]+)", - re.IGNORECASE, - ), -) - CONTEXT_WINDOW_FIELDS = ( "max_model_len", "context_length", @@ -284,31 +274,26 @@ async def chat( return await client.chat.completions.create(**kwargs) -def context_error(error: APIStatusError) -> tuple[bool, int | None]: +class CompactionFailed(Exception): + """Every checkpoint attempt failed - the caller ends the run cleanly instead.""" + + +def is_context_overflow(error: APIStatusError) -> bool: details = f"{error} {error.body or ''}" # An overflow is deterministic: a 400, or a 413 for a byte-size cap. - overflow = error.status_code in (400, 413) and any( + return error.status_code in (400, 413) and any( marker in details.casefold() for marker in CONTEXT_OVERFLOW_MARKERS ) - for pattern in CONTEXT_WINDOW_PATTERNS: - match = pattern.search(details) - if match: - context_window = int(match.group(1).replace(",", "")) - return overflow, default_threshold(context_window) - return overflow, None - - -def drop_latest_tool_result(messages: list[dict]) -> bool: - """Replace one tool result so the checkpoint request can fit in context.""" - for index in range(len(messages) - 1, -1, -1): - message = messages[index] - if message.get("role") != "tool": - continue - if message.get("content") == COMPACTED_TOOL_RESULT: - continue - messages[index] = {**message, "content": COMPACTED_TOOL_RESULT} - return True - return False + + +def compactable(messages: list[dict]) -> bool: + """Whether compaction can reclaim anything - some history beyond the task exists.""" + first_user = next( + (i for i, m in enumerate(messages) if m.get("role") == "user"), None + ) + return any( + m.get("role") != "system" and i != first_user for i, m in enumerate(messages) + ) def truncate_tool_output(text: str) -> str: @@ -347,6 +332,9 @@ def __init__(self, client, model, tools, enabled, threshold): self.tools = tools self.enabled = enabled self.threshold = threshold + self.last_good = 0 + """Message count of the newest state that passed a threshold check - by + definition a state with a full reserve of room, so a checkpoint over it fits.""" def reached(self, completion, extra_tokens: int = 0) -> bool: return ( @@ -355,18 +343,26 @@ def reached(self, completion, extra_tokens: int = 0) -> bool: and context_tokens(completion) + extra_tokens >= self.threshold ) + def note_good(self, messages: list[dict]) -> None: + self.last_good = len(messages) + async def complete(self, messages: list[dict]): try: completion = await chat(self.client, self.model, messages, self.tools) except APIStatusError as error: - overflow, threshold = context_error(error) - if not self.enabled or not overflow: + if ( + not self.enabled + or self.threshold is None + or not is_context_overflow(error) + or not compactable(messages) + ): raise - if self.threshold is None: - self.threshold = threshold else: choice = completion.choices[0] if choice.finish_reason != "length" or not self.reached(completion): + self.note_good(messages) + return completion, messages + if not compactable(messages): return completion, messages messages = await self.compact(messages) @@ -374,10 +370,14 @@ async def complete(self, messages: list[dict]): return completion, messages async def compact(self, messages: list[dict]) -> list[dict]: + # A rejected checkpoint falls back to the last good snapshot (which has a + # full reserve of room, so it fits); an empty or tool-calling reply is + # resampled. Reasoning is never part of the summary. system = [message for message in messages if message.get("role") == "system"] - while True: + base = messages + for _ in range(COMPACTION_ATTEMPTS): checkpoint = [ - *messages, + *base, {"role": "user", "content": CHECKPOINT_COMPACTION_PROMPT}, ] try: @@ -388,12 +388,20 @@ async def compact(self, messages: list[dict]) -> list[dict]: self.tools, tool_choice="none", ) - summary = completion.choices[0].message.content or "" - framed = POST_COMPACTION_FRAMING + "\n\n" + summary - return [*system, {"role": "user", "content": framed}] except APIStatusError as error: - if not context_error(error)[0] or not drop_latest_tool_result(messages): + if not is_context_overflow(error): raise + base = messages[: self.last_good] + continue + message = completion.choices[0].message + if not message.tool_calls and (message.content or "").strip(): + framed = POST_COMPACTION_FRAMING + "\n\n" + message.content + rebuilt = [*system, {"role": "user", "content": framed}] + self.note_good(rebuilt) + return rebuilt + raise CompactionFailed( + f"no usable summary after {COMPACTION_ATTEMPTS} attempts" + ) @asynccontextmanager @@ -594,7 +602,12 @@ async def main() -> None: if compactor.enabled and compactor.threshold is None: compactor.threshold = await discover_threshold(client, args.model) while True: - completion, messages = await compactor.complete(messages) + try: + completion, messages = await compactor.complete(messages) + except CompactionFailed: + # The context is exhausted and could not be summarized: end the run + # cleanly with what the conversation holds - still a trainable sample. + break choice = completion.choices[0] message = choice.message messages.append(message.model_dump(exclude_none=True)) @@ -669,8 +682,13 @@ async def main() -> None: tool_message = decision["message"] messages.append(tool_message) tool_result_tokens += estimated_tokens(str(tool_message["content"])) - if compactor.reached(completion, tool_result_tokens): - messages = await compactor.compact(messages) + if compactor.reached(completion, tool_result_tokens) and compactable(messages): + try: + messages = await compactor.compact(messages) + except CompactionFailed: + break + else: + compactor.note_good(messages) if tool_client is not None: await tool_client.aclose() diff --git a/verifiers/v1/harnesses/rlm/harness.py b/verifiers/v1/harnesses/rlm/harness.py index f7d8950d94..2e670d23a0 100644 --- a/verifiers/v1/harnesses/rlm/harness.py +++ b/verifiers/v1/harnesses/rlm/harness.py @@ -43,7 +43,7 @@ class CompactionConfig(BaseConfig): class RLMHarnessConfig(HarnessConfig): - version: str = Field(default="4bb5f48", min_length=1) + version: str = Field(default="b928097", min_length=1) """Git ref (branch, tag, or commit) of nano-rlm to install.""" max_depth: int = 0 """Recursion depth RLM may spawn sub-harnesses to.""" From 978e65012bb67f75f89a35c521c6ce4cf1b26092 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Fri, 28 Aug 2026 22:01:37 +0000 Subject: [PATCH 07/16] restrict this PR to the bash harness The RLM harness wiring moves to a stacked PR so this one can merge before the nano-rlm companion lands. Also discover the context window via models.list - the raw cast_to parse breaks on one Python version or another (a bare dict cannot be constructed on 3.13, and a parameterized dict trips inspect.isclass on 3.10). Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/program.py | 16 +++++---- verifiers/v1/harnesses/rlm/harness.py | 49 ++++++++++++++++---------- 2 files changed, 40 insertions(+), 25 deletions(-) diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index b45bec9df4..0af953a088 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -10,7 +10,6 @@ import subprocess from contextlib import AsyncExitStack, asynccontextmanager, suppress from pathlib import Path -from typing import Any import httpx from openai import APIError, APIStatusError, AsyncOpenAI @@ -87,17 +86,20 @@ def default_threshold(context_window: int) -> int: async def discover_threshold(client: AsyncOpenAI, model: str) -> int | None: - """The compaction threshold, when the provider's model card advertises a context window.""" + """The compaction threshold, when the provider's model card advertises a context window. + + `models.list()` keeps provider extensions in each card's `model_extra`; a raw + `cast_to` parse breaks on one Python version or another.""" try: - # The SDK needs a parameterized mapping type to parse into (bare `dict` fails). - payload = await client.get("/models", cast_to=dict[str, Any]) + page = await client.models.list() except APIError: return None - for card in payload.get("data") or []: - if not isinstance(card, dict) or card.get("id") != model: + for card in page.data: + if card.id != model: continue + extra = card.model_extra or {} for field in CONTEXT_WINDOW_FIELDS: - value = card.get(field) + value = extra.get(field) if isinstance(value, int) and not isinstance(value, bool) and value > 0: return default_threshold(value) break diff --git a/verifiers/v1/harnesses/rlm/harness.py b/verifiers/v1/harnesses/rlm/harness.py index 2e670d23a0..639d556c2e 100644 --- a/verifiers/v1/harnesses/rlm/harness.py +++ b/verifiers/v1/harnesses/rlm/harness.py @@ -1,11 +1,11 @@ """RLM over ACP, with MCP tools exposed as pre-imported IPython skills.""" import logging +import random import shlex from typing import Literal from pydantic import BaseModel, ConfigDict, Field, PositiveInt, model_validator -from pydantic_config import BaseConfig from verifiers.v1.acp import ACPConfig, ACPHarness, ACPTurn, JsonObject from verifiers.v1.clients import ModelContext @@ -34,24 +34,30 @@ class _SessionSnapshot(BaseModel): metrics: dict[str, int | float] -class CompactionConfig(BaseConfig): - """Context compaction policy for the RLM agent loop.""" - - summarize_at_tokens: PositiveInt | None = None - """Compact at this token count. When unset, compact when 16k tokens remain below the - model context window when the provider advertises it.""" - - class RLMHarnessConfig(HarnessConfig): - version: str = Field(default="b928097", min_length=1) + version: str = Field( + default="d4ce3e10e63b359f4f3d432d58a77471e9e21fe7", min_length=1 + ) """Git ref (branch, tag, or commit) of nano-rlm to install.""" max_depth: int = 0 """Recursion depth RLM may spawn sub-harnesses to.""" builtin_skills: list[BuiltinSkill] = Field(default_factory=list) """Built-in rlm skills to enable (RLM_SKILLS), e.g. `["edit"]`; empty enables none. The tool set is fixed (ipython); the base `skills` field takes SKILL.md paths.""" - compaction: CompactionConfig | None = None - """Context compaction policy. Set an empty config to use automatic thresholds.""" + summarize_at_tokens: PositiveInt | tuple[PositiveInt, PositiveInt] | None = None + """Auto-compaction threshold (RLM_SUMMARIZE_AT_TOKENS): compact the context once it grows + past this many tokens. An int is a fixed threshold; a `(lo, hi)` pair draws a per-group + threshold (seeded by the task index, so a task's rollouts share one draw and tasks vary). + `None` disables auto-compaction; ints must be positive.""" + + @model_validator(mode="after") + def validate_range(self) -> "RLMHarnessConfig": + value = self.summarize_at_tokens + if isinstance(value, tuple) and value[0] > value[1]: + raise ValueError( + "`summarize_at_tokens` range must be (lo, hi) with lo <= hi." + ) + return self @model_validator(mode="after") def reject_disabled_tools(self) -> "RLMHarnessConfig": @@ -91,6 +97,16 @@ async def setup(self, runtime: Runtime) -> None: raise RuntimeError(f"rlm install failed: {result.stderr.strip()[-500:]}") await super().setup(runtime) + def summarize_threshold(self, task_idx: int | None) -> int | None: + """Resolve a fixed or per-task compaction threshold.""" + value = self.config.summarize_at_tokens + if value is None: + return None + if isinstance(value, tuple): + lo, hi = value + return random.Random(task_idx or 0).randint(lo, hi) + return value + def _runtime_metadata( self, ctx: ModelContext, @@ -98,9 +114,9 @@ def _runtime_metadata( runtime: Runtime, endpoint: str, secret: str, + data: TaskData, system_prompt: str | None, ) -> JsonObject: - compaction = self.config.compaction payload = { "session_id": trace.id, "model": ctx.model, @@ -110,10 +126,7 @@ def _runtime_metadata( }, "policy": { "max_depth": self.config.max_depth, - "compaction": compaction is not None, - "summarize_at_tokens": ( - compaction.summarize_at_tokens if compaction else None - ), + "summarize_at_tokens": self.summarize_threshold(data.idx), "max_concurrent_subagents": max(4, self.config.max_depth), }, "system_prompt_path": None, @@ -140,7 +153,7 @@ async def prepare_acp( command=[RLM_BIN, "--acp"], prompt=prompt, session_meta=self._runtime_metadata( - ctx, trace, runtime, endpoint, secret, system_prompt + ctx, trace, runtime, endpoint, secret, data, system_prompt ), ) From 71d628ec58f320575a49454bfc3489bbdd4bacdb Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Fri, 28 Aug 2026 22:05:39 +0000 Subject: [PATCH 08/16] accept a summary from the reasoning channel A reasoning-parsed model (observed: Laguna via the glm45 parser) can put the entire checkpoint reply in reasoning_content, leaving content empty - every attempt then fails and the run ends as compaction-failed despite a perfectly good summary. The checkpoint asked for a summary, so when content is empty accept the reasoning text as the summary. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/program.py | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index 0af953a088..680e210464 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -396,8 +396,14 @@ async def compact(self, messages: list[dict]) -> list[dict]: base = messages[: self.last_good] continue message = completion.choices[0].message - if not message.tool_calls and (message.content or "").strip(): - framed = POST_COMPACTION_FRAMING + "\n\n" + message.content + # A reasoning-parsed model can put the whole reply in the reasoning + # channel; the checkpoint asked for a summary, so accept it from + # there when content is empty. + text = (message.content or "").strip() or ( + getattr(message, "reasoning_content", None) or "" + ).strip() + if not message.tool_calls and text: + framed = POST_COMPACTION_FRAMING + "\n\n" + text rebuilt = [*system, {"role": "user", "content": framed}] self.note_good(rebuilt) return rebuilt From f07f60ab346457620d56716cb337995a717237b8 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Fri, 28 Aug 2026 22:16:50 +0000 Subject: [PATCH 09/16] read the reasoning channel from model extras vLLM 0.26 names the field "reasoning" and the SDK only keeps it in model_extra, so the attribute lookup never saw it. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/program.py | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index 680e210464..9ecef1c5df 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -397,11 +397,14 @@ async def compact(self, messages: list[dict]) -> list[dict]: continue message = completion.choices[0].message # A reasoning-parsed model can put the whole reply in the reasoning - # channel; the checkpoint asked for a summary, so accept it from - # there when content is empty. - text = (message.content or "").strip() or ( - getattr(message, "reasoning_content", None) or "" - ).strip() + # channel ("reasoning_content" or vLLM's "reasoning"); the checkpoint + # asked for a summary, so accept it from there when content is empty. + extra = getattr(message, "model_extra", None) or {} + text = ( + (message.content or "").strip() + or str(extra.get("reasoning_content") or "").strip() + or str(extra.get("reasoning") or "").strip() + ) if not message.tool_calls and text: framed = POST_COMPACTION_FRAMING + "\n\n" + text rebuilt = [*system, {"role": "user", "content": framed}] From 84eae301ecee29903f29344eec470bb7cd3e2b53 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Fri, 28 Aug 2026 22:45:09 +0000 Subject: [PATCH 10/16] summaries use only non-reasoning output A checkpoint reply that lives entirely in the reasoning channel is resampled like an empty one - reasoning never enters the summary. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/program.py | 13 ++++--------- 1 file changed, 4 insertions(+), 9 deletions(-) diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index 9ecef1c5df..89789bea80 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -396,15 +396,10 @@ async def compact(self, messages: list[dict]) -> list[dict]: base = messages[: self.last_good] continue message = completion.choices[0].message - # A reasoning-parsed model can put the whole reply in the reasoning - # channel ("reasoning_content" or vLLM's "reasoning"); the checkpoint - # asked for a summary, so accept it from there when content is empty. - extra = getattr(message, "model_extra", None) or {} - text = ( - (message.content or "").strip() - or str(extra.get("reasoning_content") or "").strip() - or str(extra.get("reasoning") or "").strip() - ) + # Reasoning never enters the summary: only the reply's final text + # counts, so a reply that lives entirely in the reasoning channel + # is resampled like an empty one. + text = (message.content or "").strip() if not message.tool_calls and text: framed = POST_COMPACTION_FRAMING + "\n\n" + text rebuilt = [*system, {"role": "user", "content": framed}] From d9bd5155d0d0fdf05accba0bfa0ca22c51288c52 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Fri, 28 Aug 2026 23:21:50 +0000 Subject: [PATCH 11/16] harden first-turn and multimodal compaction paths Review: last_good started at zero, so a first-turn checkpoint rejection retried over an empty base - a summary of nothing with the task gone; the initial conversation is now the floor. And a multimodal MCP result is a content-part list, which the byte truncation crashed on - only plain text is truncated. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/program.py | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index 89789bea80..2a8359cf6c 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -607,6 +607,9 @@ async def main() -> None: ) if compactor.enabled and compactor.threshold is None: compactor.threshold = await discover_threshold(client, args.model) + # The initial conversation is the floor for checkpoint fallbacks: a first-turn + # checkpoint must never retry from an empty base. + compactor.note_good(messages) while True: try: completion, messages = await compactor.complete(messages) @@ -674,7 +677,11 @@ async def main() -> None: ) else: content = f"error: unknown tool {name!r}" - tool_message["content"] = truncate_tool_output(content) + # Multimodal MCP results come back as content-part lists; only plain + # text is truncated. + tool_message["content"] = ( + truncate_tool_output(content) if isinstance(content, str) else content + ) if args.tool_interception_url: assert tool_client is not None decision = await run_tool_hook( From 59bfc657f71fef37b7df7b7c99d31b0864fb33ba Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Fri, 28 Aug 2026 23:32:24 +0000 Subject: [PATCH 12/16] bound rewritten tool messages too Review: interception hook rewrites entered the conversation unbounded, sidestepping the 20KB tool-output limit. Every message entering as a tool result now passes the same bound, rewrites included. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/program.py | 20 +++++++++++++------- 1 file changed, 13 insertions(+), 7 deletions(-) diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index 2a8359cf6c..53f86f3afe 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -298,6 +298,15 @@ def compactable(messages: list[dict]) -> bool: ) +def bound_tool_message(message: dict) -> dict: + """Bound a tool message before it enters the conversation - rewrites included.""" + content = message.get("content") + if isinstance(content, str): + return {**message, "content": truncate_tool_output(content)} + # Multimodal results come back as content-part lists; only plain text is truncated. + return message + + def truncate_tool_output(text: str) -> str: """Keep the head and tail of an oversized tool result and say what was cut.""" data = text.encode("utf-8") @@ -641,7 +650,7 @@ async def main() -> None: tool_message, ) if decision["action"] == "rewrite": - rewritten = decision["message"] + rewritten = bound_tool_message(decision["message"]) messages.append(rewritten) tool_result_tokens += estimated_tokens( str(rewritten.get("content", "")) @@ -677,11 +686,8 @@ async def main() -> None: ) else: content = f"error: unknown tool {name!r}" - # Multimodal MCP results come back as content-part lists; only plain - # text is truncated. - tool_message["content"] = ( - truncate_tool_output(content) if isinstance(content, str) else content - ) + tool_message["content"] = content + tool_message = bound_tool_message(tool_message) if args.tool_interception_url: assert tool_client is not None decision = await run_tool_hook( @@ -692,7 +698,7 @@ async def main() -> None: tool_message, ) if decision["action"] == "rewrite": - tool_message = decision["message"] + tool_message = bound_tool_message(decision["message"]) messages.append(tool_message) tool_result_tokens += estimated_tokens(str(tool_message["content"])) if compactor.reached(completion, tool_result_tokens) and compactable(messages): From 7c05e5c25874d51459e5ce31096014b118ae1645 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Fri, 28 Aug 2026 23:37:08 +0000 Subject: [PATCH 13/16] end the run cleanly when the retry still overflows Review: an overflow on the post-compaction work call propagated out of the loop and crashed the rollout. The rebuilt conversation is sized to fit by construction, so if it still overflows there are no moves left - convert it to the compaction-failed clean ending. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/program.py | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index 53f86f3afe..d2eb9f6429 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -377,7 +377,15 @@ async def complete(self, messages: list[dict]): return completion, messages messages = await self.compact(messages) - completion = await chat(self.client, self.model, messages, self.tools) + try: + completion = await chat(self.client, self.model, messages, self.tools) + except APIStatusError as error: + # The rebuilt conversation is sized to fit, so this is out of moves. + if is_context_overflow(error): + raise CompactionFailed( + "the rebuilt conversation still overflows" + ) from error + raise return completion, messages async def compact(self, messages: list[dict]) -> list[dict]: From cad252f97872538eae42589bbab2d7c16f55eec4 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Fri, 28 Aug 2026 23:45:19 +0000 Subject: [PATCH 14/16] only usage-verified states become checkpoint fallbacks Review: the post-tool snapshot was taken on a chars/4 estimate, which can undercount dense content severalfold - the "good" snapshot could itself be oversized, making the fallback identical to the overflowing request. A state now becomes the fallback only when the provider accepted that exact prompt with real usage below the threshold, which lands the fallback before the tool results, as designed. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/bash/program.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/verifiers/v1/harnesses/bash/program.py b/verifiers/v1/harnesses/bash/program.py index d2eb9f6429..d4d91e7f21 100644 --- a/verifiers/v1/harnesses/bash/program.py +++ b/verifiers/v1/harnesses/bash/program.py @@ -370,10 +370,12 @@ async def complete(self, messages: list[dict]): raise else: choice = completion.choices[0] - if choice.finish_reason != "length" or not self.reached(completion): + if not self.reached(completion): + # Usage-verified: this exact prompt was accepted with a full + # reserve of room, so it is a safe checkpoint fallback. self.note_good(messages) return completion, messages - if not compactable(messages): + if choice.finish_reason != "length" or not compactable(messages): return completion, messages messages = await self.compact(messages) @@ -714,8 +716,6 @@ async def main() -> None: messages = await compactor.compact(messages) except CompactionFailed: break - else: - compactor.note_good(messages) if tool_client is not None: await tool_client.aclose() From 27e1c91eebd6d911be08e81434ac7222ec9cac01 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Mon, 31 Aug 2026 19:40:51 +0000 Subject: [PATCH 15/16] end cleanly when a compaction floor still overflows An overflow on a conversation that a compaction already reduced to [system, summary] re-raised the provider error, so the rollout failed instead of ending as a trainable sample like the in-cycle retry path. Track that a compaction happened and convert that overflow to CompactionFailed; the first-turn floor keeps raising. Also give the /v1/models relay a finite read timeout so a hung provider cannot stall threshold discovery for the rollout's whole outer timeout. Co-Authored-By: Claude Fable 5 --- verifiers/v1/harnesses/minimal/program.py | 11 ++++++++++- verifiers/v1/interception/server.py | 8 ++++++-- 2 files changed, 16 insertions(+), 3 deletions(-) diff --git a/verifiers/v1/harnesses/minimal/program.py b/verifiers/v1/harnesses/minimal/program.py index 6376dbb745..6535458f9a 100644 --- a/verifiers/v1/harnesses/minimal/program.py +++ b/verifiers/v1/harnesses/minimal/program.py @@ -355,6 +355,7 @@ def __init__(self, client, model, tools, enabled, threshold): self.tools = tools self.enabled = enabled self.threshold = threshold + self.compacted = False self.last_good = 0 """Message count of the newest state that passed a threshold check - by definition a state with a full reserve of room, so a checkpoint over it fits.""" @@ -377,9 +378,16 @@ async def complete(self, messages: list[dict]): not self.enabled or self.threshold is None or not is_context_overflow(error) - or not compactable(messages) ): raise + if not compactable(messages): + if self.compacted: + # The conversation is already a compaction floor and still + # overflows - out of moves, end cleanly. + raise CompactionFailed( + "the compacted conversation still overflows" + ) from error + raise else: choice = completion.choices[0] if not self.reached(completion): @@ -435,6 +443,7 @@ async def compact(self, messages: list[dict]) -> list[dict]: framed = POST_COMPACTION_FRAMING + "\n\n" + text rebuilt = [*system, {"role": "user", "content": framed}] self.note_good(rebuilt) + self.compacted = True return rebuilt raise CompactionFailed( f"no usable summary after {COMPACTION_ATTEMPTS} attempts" diff --git a/verifiers/v1/interception/server.py b/verifiers/v1/interception/server.py index 8731cbdadf..64ac6dbdda 100644 --- a/verifiers/v1/interception/server.py +++ b/verifiers/v1/interception/server.py @@ -40,7 +40,7 @@ from verifiers.v1 import graph from verifiers.v1.clients import Client, resolve_client -from verifiers.v1.clients.base import DEFAULT_TIMEOUT, join_url +from verifiers.v1.clients.base import join_url from verifiers.v1.configs.client import BaseClientConfig, resolve_api_key from verifiers.v1.dialects import DIALECTS, Dialect from verifiers.v1.dialects.base import ( @@ -1076,7 +1076,11 @@ async def handle_models(self, request: web.Request) -> web.Response: headers = dict(config.headers or {}) headers.update(dialect.auth_headers(resolve_api_key(config))) try: - async with httpx.AsyncClient(timeout=DEFAULT_TIMEOUT) as client: + # Finite read timeout: a hung provider must not stall threshold discovery + # for the rollout's whole outer timeout - the loop falls back to no compaction. + async with httpx.AsyncClient( + timeout=httpx.Timeout(30.0, connect=5.0) + ) as client: upstream = await client.get( join_url(config.base_url, "/v1/models"), headers=headers ) From e1c80ac223f7d63784ad1e38e25d229f21826b53 Mon Sep 17 00:00:00 2001 From: Mika Senghaas Date: Mon, 31 Aug 2026 20:36:55 +0000 Subject: [PATCH 16/16] drop a route-registration comment Co-Authored-By: Claude Fable 5 --- verifiers/v1/interception/server.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/verifiers/v1/interception/server.py b/verifiers/v1/interception/server.py index 64ac6dbdda..258d21170e 100644 --- a/verifiers/v1/interception/server.py +++ b/verifiers/v1/interception/server.py @@ -328,8 +328,6 @@ async def start(self) -> None: app.router.add_post(route, self._handler_for(dialect)) for aux in dialect.aux_routes: app.router.add_post(aux, self._aux_handler_for(dialect, aux)) - # One models route serves every dialect: OpenAI and Anthropic SDKs both list - # models at `GET /v1/models` (the response schema is the upstream's). app.router.add_get("/v1/models", self.handle_models) # Tool servers use a state-only capability; the model bearer cannot reach these. app.router.add_get("/state", self.handle_state_get)