From 4d2477942f7caa48106a3619c549b2fb1b12d0e1 Mon Sep 17 00:00:00 2001 From: Kyzcreig <9063726+Kyzcreig@users.noreply.github.com> Date: Fri, 25 Sep 2026 15:56:16 -0700 Subject: [PATCH] revert(budget): drop grace-turn side-effect lockout (#37) (t_0973b125) Fork-PR audit DROP (lead t_03e35f0e, FINAL.md docs PR #1128): 0 grace-turn denials in 555,617 turn_tool_calls (06-11..09-25). - delete agent/budget_grace_gate.py + tests/agent/test_budget_grace_gate.py - cut the grace branch from agent/fork_ext/tool_gate.py pre_tool_block_from_builtin_gate (scope block unchanged) - drop _in_budget_grace from agent_init / conversation_loop; the upstream _budget_grace_call grace turn itself is untouched - tool_executor: drop grace_block_result special case in the block ladder - tests/golden/tool_gate: remove the 2 grace=true cases and the _in_budget_grace agent attrs from corpus, runner adapter branch removed, golden.json regenerated (10 cases) Registry D2b entry 11 (tool_gate): paths agent/fork_ext/tool_gate.py and tests/golden/tool_gate/ still exist; no registry edit needed. Verified: refactor_equiv verify tool_gate golden rc=0; narrow pytest (guardrail runtime, tool_search scoping, TestBudgetPressure, delegate_toolset_scope, test_refactor_equiv) 27 passed via test-gate. --- agent/agent_init.py | 3 - agent/budget_grace_gate.py | 114 --------- agent/conversation_loop.py | 12 - agent/fork_ext/tool_gate.py | 7 - agent/tool_executor.py | 27 +-- tests/agent/test_budget_grace_gate.py | 317 -------------------------- tests/golden/tool_gate/corpus.json | 20 +- tests/golden/tool_gate/golden.json | 87 ++----- tests/golden/tool_gate/runner.py | 7 - 9 files changed, 31 insertions(+), 563 deletions(-) delete mode 100644 agent/budget_grace_gate.py delete mode 100644 tests/agent/test_budget_grace_gate.py diff --git a/agent/agent_init.py b/agent/agent_init.py index e94f21eb9ec8..e8d66089cdde 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -1084,9 +1084,6 @@ def init_agent( # models to "give up" prematurely on complex tasks (#7915). agent._budget_exhausted_injected = False agent._budget_grace_call = False - # True only during the one post-budget grace turn; read by the tool - # dispatchers to refuse side-effecting tools then (Guard D-core). - agent._in_budget_grace = False # Optional wall-clock run budget (seconds per run_conversation turn). # Explicit constructor arg wins; else resolved from config.yaml diff --git a/agent/budget_grace_gate.py b/agent/budget_grace_gate.py deleted file mode 100644 index 64c09621420c..000000000000 --- a/agent/budget_grace_gate.py +++ /dev/null @@ -1,114 +0,0 @@ -"""Budget-grace tool gate — deny-by-default side-effect lockout for the grace turn. - -Background ----------- -The tool-calling loop in ``agent/conversation_loop.py`` runs while the agent has -iteration budget remaining, plus **one optional grace turn** after the budget is -exhausted (the ``agent._budget_grace_call`` flag). The grace turn exists so the -model can write a clean final summary after it runs out of budget. - -Today the grace turn would execute *whatever* tools the model returns — including -side-effecting ones (``terminal``, ``execute_code``, ``write_file``, -``delegate_task``, ``send_message``, ...). For a runaway worker that is one more -chance to spawn a subprocess or write to disk *after* its budget is already gone. - -This module is the gate: during the grace turn, allow only an explicit read-only -allowlist (the reads a model needs to compose a final answer) and **refuse every -other tool — deny-by-default**, including any unrecognized / future / third-party -tool name. It is the true root-cause complement to the runaway-worker guards -(B/E/F) which cap blast radius *around* the loop; this stops the loop itself from -taking a side-effecting action past its budget. - -Design notes ------------- -- **Deny-by-default, NOT a denylist.** A new or unknown tool name is refused. The - allowlist is a small, explicit, audited set of read-only finalization tools. -- **Pure / side-effect free.** Mirrors ``agent/tool_guardrails.py``: this module - only decides; the runtime owns turning a decision into a synthetic tool result. -- **Mutating-tool overlap guard.** As defense-in-depth, any name that the existing - guardrails consider mutating is refused even if (by mistake) it were ever added - to the allowlist — the mutating set always wins. -""" - -from __future__ import annotations - -import json - -from agent.tool_guardrails import MUTATING_TOOL_NAMES - - -# Read-only tools the model may still call during the grace turn so it can -# gather context and write a clean final summary. Deny-by-default: anything not -# in this set (including unknown/future tools) is refused. -# -# Kept deliberately small and audited. Every entry must be genuinely read-only -# (no disk writes, no subprocess, no network mutation, no message send, no -# subagent spawn). ``memory``/``todo`` are intentionally EXCLUDED: they mutate -# state and are not needed to compose a final answer. -GRACE_READONLY_TOOLS = frozenset( - { - "read_file", - "search_files", - "TaskSearch", - "session_search", - "mem0_search", - "mem0_profile", - "skill_view", - "skills_list", - # read-only MCP filesystem tools (mirror tool_guardrails idempotent set) - "mcp_filesystem_read_file", - "mcp_filesystem_read_text_file", - "mcp_filesystem_read_multiple_files", - "mcp_filesystem_list_directory", - "mcp_filesystem_list_directory_with_sizes", - "mcp_filesystem_directory_tree", - "mcp_filesystem_get_file_info", - "mcp_filesystem_search_files", - } -) - - -def is_readonly_grace_tool(tool_name: str) -> bool: - """Return True iff *tool_name* may execute during the budget grace turn. - - Deny-by-default: only the explicit read-only allowlist returns True. The - mutating set always wins — a name that is both allowlisted (by mistake) and - mutating is still refused — so the allowlist can never re-open a side effect. - """ - if not isinstance(tool_name, str) or not tool_name: - return False - if tool_name in MUTATING_TOOL_NAMES: - return False - return tool_name in GRACE_READONLY_TOOLS - - -def grace_block_message(tool_name: str) -> str: - """Human-readable reason a tool was refused during the grace turn.""" - return ( - f"Refused '{tool_name}': the iteration budget is exhausted and this is the " - "final grace turn. Only read-only tools may run now. Write your final " - "summary/answer directly instead of calling side-effecting tools." - ) - - -def grace_block_result(tool_name: str) -> str: - """Build the synthetic role=tool content string for a grace-refused call. - - Shape mirrors the plugin/guardrail block path in the dispatchers - (``{"error": ...}``) so the model sees a consistent blocked-tool result. - """ - return json.dumps( - { - "error": grace_block_message(tool_name), - "budget_grace_block": {"tool_name": tool_name, "reason": "budget_exhausted_grace_turn"}, - }, - ensure_ascii=False, - ) - - -__all__ = [ - "GRACE_READONLY_TOOLS", - "is_readonly_grace_tool", - "grace_block_message", - "grace_block_result", -] diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index c956fbaf833a..ae180909938c 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -2652,20 +2652,8 @@ def run_conversation( # Grace call: the budget is exhausted but we gave the model one # more chance. Consume the grace flag so the loop exits after # this iteration regardless of outcome. - # - # ``_in_budget_grace`` is recomputed every iteration (default False) and - # set True ONLY for the grace turn. The tool dispatchers read it to - # refuse side-effecting tools during the grace turn (deny-by-default, - # read-only allowlist) — see ``agent/budget_grace_gate.py`` (Guard - # D-core). Refusing a call must NOT re-arm ``_budget_grace_call``: the - # flag is already cleared here, so the loop exits after this iteration - # whether the model's tool calls execute or are refused. A - # deny-that-loops would itself be a runaway, so the gate only blocks - # execution; it never extends the loop. - agent._in_budget_grace = False if agent._budget_grace_call: agent._budget_grace_call = False - agent._in_budget_grace = True elif not agent.iteration_budget.consume(): _turn_exit_reason = "budget_exhausted" if not agent.quiet_mode: diff --git a/agent/fork_ext/tool_gate.py b/agent/fork_ext/tool_gate.py index cd24eb9c811c..306aa2f41c62 100644 --- a/agent/fork_ext/tool_gate.py +++ b/agent/fork_ext/tool_gate.py @@ -5,8 +5,6 @@ import json from typing import Any -from agent.budget_grace_gate import grace_block_message, is_readonly_grace_tool - def tool_search_scoped_names(agent) -> frozenset: try: @@ -80,11 +78,6 @@ def resolve_tool_search_unwrap(agent, function_name: str, function_args: dict[st def pre_tool_block_from_builtin_gate(agent, function_name: str, tool_scope_block: str | None) -> dict[str, str] | None: - if getattr(agent, "_in_budget_grace", False) and not is_readonly_grace_tool(function_name): - return { - "message": grace_block_message(function_name), - "error_type": "budget_grace_block", - } if tool_scope_block is not None: return { "message": tool_scope_block, diff --git a/agent/tool_executor.py b/agent/tool_executor.py index 0e294b02d4f9..d4aaed833cc8 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -33,9 +33,6 @@ _detect_tool_failure, ) from agent.tool_guardrails import ToolGuardrailDecision -from agent.budget_grace_gate import ( - grace_block_result, -) from agent.fork_ext.tool_gate import ( pre_tool_block_from_builtin_gate, resolve_tool_search_unwrap, @@ -783,13 +780,8 @@ def _advance_start_order(callback=None) -> None: begin_execution(callback) # ── Block ladder ──────────────────────────────────────────────── - # parity NOTE (upstream->fork merge 2026-08-08): the fork-only - # budget-grace gate (Guard D-core) used to live in each dispatcher. - # Upstream moved plugin/scope/guardrail blocking here, so the gate is - # re-grafted at the head of THIS ladder — the single choke point both - # dispatchers funnel through — rather than duplicated per path. It - # runs FIRST so an exhausted-budget turn is refused before any plugin - # hook or guardrail bookkeeping observes the call. + # Fork tool-search scope block first (agent/fork_ext/tool_gate.py), + # then plugin pre_tool_call hooks, then guardrails. block_message = pre_tool_block_from_builtin_gate( agent, function_name, scope_block ) @@ -841,18 +833,9 @@ def _resolve_pre_tool_block(): _advance_start_order() state["blocked"] = True if block_message is not None: - # parity NOTE (upstream->fork merge 2026-08-08): the fork's - # budget-grace refusal must carry the budget_grace_block - # metadata key (asserted by test_block_message_and_result_shape - # and test_grace_block_result_metadata_key_present_in_real_dispatch), - # so it uses the shared grace_block_result() rather than the - # generic {"error": ...} envelope every other block emits. - if block_error_type == "budget_grace_block": - result = grace_block_result(function_name) - else: - result = json.dumps( - {"error": block_message}, ensure_ascii=False - ) + result = json.dumps( + {"error": block_message}, ensure_ascii=False + ) error_type = block_error_type error_message = block_message else: diff --git a/tests/agent/test_budget_grace_gate.py b/tests/agent/test_budget_grace_gate.py deleted file mode 100644 index 755eca31f1f6..000000000000 --- a/tests/agent/test_budget_grace_gate.py +++ /dev/null @@ -1,317 +0,0 @@ -"""Guard D-core — budget-grace tool gate tests. - -Two layers: - -1. **Pure predicate** (``agent/budget_grace_gate.py``): allowlist / deny-by-default - / mutating-overlap / block-message shape. -2. **Real dispatch integration**: drive the actual sequential AND concurrent tool - executors with ``agent._in_budget_grace = True`` and assert side-effecting - tools are refused (``handle_function_call`` never called, a blocked role=tool - message is appended) while read-only tools still execute. A mixed batch - executes the read and refuses the side effect (per-call gating). - -Why integration and not just the loop end-to-end: ``_budget_grace_call`` is a -dormant hook (never armed to True today), so a "real bounded task that exhausts -budget" exits *before* any grace turn and would prove nothing. Setting -``_in_budget_grace = True`` directly exercises the exact state the loop sets on -the grace turn (``agent/conversation_loop.py``) and drives the real dispatchers. -""" - -import json -import uuid -from types import SimpleNamespace -from unittest.mock import MagicMock, patch - -from run_agent import AIAgent - -from agent.budget_grace_gate import ( - GRACE_READONLY_TOOLS, - grace_block_message, - grace_block_result, - is_readonly_grace_tool, -) -from agent.tool_guardrails import MUTATING_TOOL_NAMES - - -# ────────────────────────────────────────────────────────────────────────── -# Layer 1 — pure predicate -# ────────────────────────────────────────────────────────────────────────── - - -def test_readonly_allowlisted_tool_is_permitted(): - assert is_readonly_grace_tool("read_file") is True - assert is_readonly_grace_tool("search_files") is True - assert is_readonly_grace_tool("skill_view") is True - - -def test_side_effecting_tool_is_refused(): - for name in ("terminal", "execute_code", "write_file", "delegate_task", "send_message"): - assert is_readonly_grace_tool(name) is False, name - - -def test_unknown_or_future_tool_is_refused_deny_by_default(): - # The crux of deny-by-default: a name nobody has seen is refused. - assert is_readonly_grace_tool("some_future_tool") is False - assert is_readonly_grace_tool("third_party_mcp_doomsday") is False - assert is_readonly_grace_tool("") is False - assert is_readonly_grace_tool(None) is False # type: ignore[arg-type] - - -def test_mutating_set_always_wins_over_allowlist(): - # Defense-in-depth: no allowlist entry may also be a known mutating tool. - overlap = GRACE_READONLY_TOOLS & MUTATING_TOOL_NAMES - assert overlap == frozenset(), f"allowlist must not contain mutating tools: {overlap}" - # memory/todo mutate state and must be refused even though they're "cheap". - assert is_readonly_grace_tool("memory") is False - assert is_readonly_grace_tool("todo") is False - - -def test_block_message_and_result_shape(): - msg = grace_block_message("terminal") - assert "terminal" in msg and "budget" in msg.lower() - parsed = json.loads(grace_block_result("terminal")) - assert parsed["error"] == msg - assert parsed["budget_grace_block"]["tool_name"] == "terminal" - assert parsed["budget_grace_block"]["reason"] == "budget_exhausted_grace_turn" - - -# ────────────────────────────────────────────────────────────────────────── -# Layer 2 — real dispatch integration -# ────────────────────────────────────────────────────────────────────────── - - -def _make_tool_defs(*names: str) -> list[dict]: - return [ - { - "type": "function", - "function": { - "name": name, - "description": f"{name} tool", - "parameters": {"type": "object", "properties": {}}, - }, - } - for name in names - ] - - -def _mock_tool_call(name="read_file", arguments="{}", call_id=None): - return SimpleNamespace( - id=call_id or f"call_{uuid.uuid4().hex[:8]}", - type="function", - function=SimpleNamespace(name=name, arguments=arguments), - ) - - -def _make_agent(*tool_names: str, max_iterations: int = 10) -> AIAgent: - with ( - patch("run_agent.get_tool_definitions", return_value=_make_tool_defs(*tool_names)), - patch("run_agent.check_toolset_requirements", return_value={}), - patch("hermes_cli.config.load_config", return_value={}), - patch("run_agent.OpenAI"), - ): - agent = AIAgent( - api_key="test-key-1234567890", - base_url="https://openrouter.ai/api/v1", - max_iterations=max_iterations, - quiet_mode=True, - skip_context_files=True, - skip_memory=True, - ) - agent.client = MagicMock() - agent._cached_system_prompt = "You are helpful." - agent._use_prompt_caching = False - agent.tool_delay = 0 - agent.compression_enabled = False - agent.save_trajectories = False - return agent - - -def _tool_msgs(messages): - return [m for m in messages if isinstance(m, dict) and m.get("role") == "tool"] - - -def test_default_in_budget_grace_is_false_after_init(): - agent = _make_agent("read_file") - assert getattr(agent, "_in_budget_grace", None) is False - - -def test_grace_turn_refuses_side_effecting_tool_sequential(): - agent = _make_agent("terminal") - agent._in_budget_grace = True - tc = _mock_tool_call("terminal", json.dumps({"command": "rm -rf /"}), "c-term") - msg = SimpleNamespace(content="", tool_calls=[tc]) - messages = [] - - with patch("run_agent.handle_function_call", return_value="SHOULD_NOT_RUN") as mock_hfc: - agent._execute_tool_calls_sequential(msg, messages, "task-1") - - mock_hfc.assert_not_called() - tmsgs = _tool_msgs(messages) - assert len(tmsgs) == 1 - assert tmsgs[0]["tool_call_id"] == "c-term" - body = json.loads(tmsgs[0]["content"]) - assert "budget" in body["error"].lower() - - -def test_grace_turn_allows_readonly_tool_sequential(): - agent = _make_agent("read_file") - agent._in_budget_grace = True - tc = _mock_tool_call("read_file", json.dumps({"path": "/tmp/x"}), "c-read") - msg = SimpleNamespace(content="", tool_calls=[tc]) - messages = [] - - with patch("run_agent.handle_function_call", return_value=json.dumps({"ok": True})) as mock_hfc: - agent._execute_tool_calls_sequential(msg, messages, "task-1") - - mock_hfc.assert_called_once() - tmsgs = _tool_msgs(messages) - assert len(tmsgs) == 1 - assert tmsgs[0]["tool_call_id"] == "c-read" - assert "budget" not in tmsgs[0]["content"].lower() - - -def test_grace_turn_refuses_unknown_tool_sequential(): - agent = _make_agent("read_file") # tool not even registered - agent._in_budget_grace = True - tc = _mock_tool_call("some_future_tool", "{}", "c-unknown") - msg = SimpleNamespace(content="", tool_calls=[tc]) - messages = [] - - with patch("run_agent.handle_function_call", return_value="SHOULD_NOT_RUN") as mock_hfc: - agent._execute_tool_calls_sequential(msg, messages, "task-1") - - mock_hfc.assert_not_called() - tmsgs = _tool_msgs(messages) - assert len(tmsgs) == 1 - # Sequential path emits the shared {"error": } block shape; the message - # names the refused tool and the budget reason. - body = json.loads(tmsgs[0]["content"]) - assert "some_future_tool" in body["error"] - assert "budget" in body["error"].lower() - - -def test_grace_turn_mixed_batch_executes_read_refuses_side_effect_sequential(): - agent = _make_agent("read_file", "terminal") - agent._in_budget_grace = True - calls = [ - _mock_tool_call("read_file", json.dumps({"path": "/tmp/x"}), "c-read"), - _mock_tool_call("terminal", json.dumps({"command": "id"}), "c-term"), - ] - msg = SimpleNamespace(content="", tool_calls=calls) - messages = [] - executed = [] - - def fake_handle(name, args, task_id, **kwargs): - executed.append(name) - return json.dumps({"ok": name}) - - with patch("run_agent.handle_function_call", side_effect=fake_handle): - agent._execute_tool_calls_sequential(msg, messages, "task-1") - - # read_file executed; terminal refused (per-call gating, not all-or-nothing). - assert executed == ["read_file"] - by_id = {m["tool_call_id"]: m for m in _tool_msgs(messages)} - assert "budget" not in by_id["c-read"]["content"].lower() - term_body = json.loads(by_id["c-term"]["content"]) - assert "terminal" in term_body["error"] and "budget" in term_body["error"].lower() - - -def test_grace_turn_refuses_side_effecting_tool_concurrent(): - agent = _make_agent("terminal", "execute_code") - agent._in_budget_grace = True - calls = [ - _mock_tool_call("terminal", json.dumps({"command": "id"}), "c-term"), - _mock_tool_call("execute_code", json.dumps({"code": "x=1"}), "c-code"), - ] - msg = SimpleNamespace(content="", tool_calls=calls) - messages = [] - - with patch("run_agent.handle_function_call", return_value="SHOULD_NOT_RUN") as mock_hfc: - agent._execute_tool_calls_concurrent(msg, messages, "task-1") - - mock_hfc.assert_not_called() - tmsgs = _tool_msgs(messages) - assert {m["tool_call_id"] for m in tmsgs} == {"c-term", "c-code"} - for m in tmsgs: - assert "budget" in m["content"].lower() - - -def test_grace_turn_mixed_batch_executes_read_refuses_side_effect_concurrent(): - # Concurrent counterpart to the sequential mixed-batch test: per-call gating - # in _execute_tool_calls_concurrent — the read runs, the side effect is - # refused (not all-or-nothing). - agent = _make_agent("read_file", "terminal") - agent._in_budget_grace = True - calls = [ - _mock_tool_call("read_file", json.dumps({"path": "/tmp/x"}), "c-read"), - _mock_tool_call("terminal", json.dumps({"command": "id"}), "c-term"), - ] - msg = SimpleNamespace(content="", tool_calls=calls) - messages = [] - executed = [] - - def fake_handle(name, args, task_id, **kwargs): - executed.append(name) - return json.dumps({"ok": name}) - - with patch("run_agent.handle_function_call", side_effect=fake_handle): - agent._execute_tool_calls_concurrent(msg, messages, "task-1") - - # read_file executed; terminal refused (per-call gating in the concurrent path). - assert executed == ["read_file"] - by_id = {m["tool_call_id"]: m for m in _tool_msgs(messages)} - assert "budget" not in by_id["c-read"]["content"].lower() - term_body = json.loads(by_id["c-term"]["content"]) - assert "terminal" in term_body["error"] and "budget" in term_body["error"].lower() - - -def test_grace_block_result_metadata_key_present_in_real_dispatch(): - # The crux of the shared-helper fix: both dispatch paths must emit the - # grace_block_result() shape (with the budget_grace_block metadata key the - # unit test asserts), not a bare {"error": ...}. Proven against BOTH real - # executors so the tested shape and the runtime-emitted shape can't diverge. - for dispatch in ("sequential", "concurrent"): - agent = _make_agent("terminal") - agent._in_budget_grace = True - tc = _mock_tool_call("terminal", json.dumps({"command": "id"}), "c-term") - msg = SimpleNamespace(content="", tool_calls=[tc]) - messages = [] - with patch("run_agent.handle_function_call", return_value="SHOULD_NOT_RUN"): - getattr(agent, f"_execute_tool_calls_{dispatch}")(msg, messages, "task-1") - body = json.loads(_tool_msgs(messages)[0]["content"]) - assert body["budget_grace_block"]["tool_name"] == "terminal", dispatch - assert body["budget_grace_block"]["reason"] == "budget_exhausted_grace_turn", dispatch - - -def test_no_grace_means_normal_execution_side_effect_runs(): - # Control: with _in_budget_grace False (the normal state), a side-effecting - # tool runs as usual — the gate must not block outside the grace turn. - agent = _make_agent("terminal") - assert agent._in_budget_grace is False - tc = _mock_tool_call("terminal", json.dumps({"command": "id"}), "c-term") - msg = SimpleNamespace(content="", tool_calls=[tc]) - messages = [] - - with patch("run_agent.handle_function_call", return_value=json.dumps({"exit_code": 0})) as mock_hfc: - agent._execute_tool_calls_sequential(msg, messages, "task-1") - - mock_hfc.assert_called_once() - tmsgs = _tool_msgs(messages) - assert "budget_grace_block" not in tmsgs[0]["content"] - - -def test_grace_flag_is_not_re_armed_by_a_refusal(): - # The gate refuses execution but must NEVER set _budget_grace_call back to - # True — a deny-that-loops would itself be a runaway. Refusing a call leaves - # the grace flag cleared so the loop exits after this iteration. - agent = _make_agent("terminal") - agent._in_budget_grace = True - agent._budget_grace_call = False # loop already consumed it this turn - tc = _mock_tool_call("terminal", json.dumps({"command": "id"}), "c-term") - msg = SimpleNamespace(content="", tool_calls=[tc]) - messages = [] - - with patch("run_agent.handle_function_call", return_value="SHOULD_NOT_RUN"): - agent._execute_tool_calls_sequential(msg, messages, "task-1") - - assert agent._budget_grace_call is False # not re-armed diff --git a/tests/golden/tool_gate/corpus.json b/tests/golden/tool_gate/corpus.json index a342fdadd83b..b854fd07fa61 100644 --- a/tests/golden/tool_gate/corpus.json +++ b/tests/golden/tool_gate/corpus.json @@ -56,30 +56,16 @@ "tool_name": "golden_lookup" }, { - "name": "budget grace blocks side effect", + "name": "tool scope block surfaces", "kind": "pre_block", - "agent": {"_in_budget_grace": true}, - "function_name": "terminal", - "tool_scope_block": null - }, - { - "name": "budget grace allows readonly", - "kind": "pre_block", - "agent": {"_in_budget_grace": true}, - "function_name": "read_file", - "tool_scope_block": null - }, - { - "name": "tool scope outranks plugin path after grace", - "kind": "pre_block", - "agent": {"_in_budget_grace": false}, + "agent": {}, "function_name": "golden_lookup", "tool_scope_block": "'golden_lookup' is not available in this session. Use tool_search to find tools you can call." }, { "name": "no pre block", "kind": "pre_block", - "agent": {"_in_budget_grace": false}, + "agent": {}, "function_name": "read_file", "tool_scope_block": null }, diff --git a/tests/golden/tool_gate/golden.json b/tests/golden/tool_gate/golden.json index 1b9dfb03d1dd..c22f3645c04b 100644 --- a/tests/golden/tool_gate/golden.json +++ b/tests/golden/tool_gate/golden.json @@ -1,41 +1,4 @@ { - "0f2e825e647e7033e4e13a52ce3802327d55c7ca5af184e72318781bd4db1365": { - "input": { - "agent": { - "_in_budget_grace": false - }, - "function_name": "read_file", - "kind": "pre_block", - "name": "no pre block", - "tool_scope_block": null - }, - "name": "no pre block", - "output": { - "db": [], - "messages": [], - "return": null - } - }, - "1a1fb7de9d0abbb6382fc0ff62e3be4dbc86b38896f4c4c36184d9196ad1ee6e": { - "input": { - "agent": { - "_in_budget_grace": false - }, - "function_name": "golden_lookup", - "kind": "pre_block", - "name": "tool scope outranks plugin path after grace", - "tool_scope_block": "'golden_lookup' is not available in this session. Use tool_search to find tools you can call." - }, - "name": "tool scope outranks plugin path after grace", - "output": { - "db": [], - "messages": [], - "return": { - "error_type": "tool_scope_block", - "message": "'golden_lookup' is not available in this session. Use tool_search to find tools you can call." - } - } - }, "1c60b7f97bf3e72b7a9333ebc9a3faa8cd6cbe9f3fb362d163a142a22716a32b": { "input": { "kind": "strip_toolsets", @@ -59,6 +22,21 @@ ] } }, + "3d3fb4c6c735c9ec69823ef08c385d78fcb14cd20360ca91a979e6a4d3a11a37": { + "input": { + "agent": {}, + "function_name": "read_file", + "kind": "pre_block", + "name": "no pre block", + "tool_scope_block": null + }, + "name": "no pre block", + "output": { + "db": [], + "messages": [], + "return": null + } + }, "4425881163593b7f8fe4a767ba5ae1643a56f79d48cd08a9d8fc51482330e93c": { "input": { "kind": "scope_message", @@ -114,43 +92,24 @@ } } }, - "95d981047ca52a81ad9a5b130d56e9085143cd4076f6f1797d6d5f12d6988c3a": { + "a2ae60c8dbf0bb71ec574eb4fb55bad6ac8fa2dd872dd5f0a5f17e31a1a62a0f": { "input": { - "agent": { - "_in_budget_grace": true - }, - "function_name": "terminal", + "agent": {}, + "function_name": "golden_lookup", "kind": "pre_block", - "name": "budget grace blocks side effect", - "tool_scope_block": null + "name": "tool scope block surfaces", + "tool_scope_block": "'golden_lookup' is not available in this session. Use tool_search to find tools you can call." }, - "name": "budget grace blocks side effect", + "name": "tool scope block surfaces", "output": { "db": [], "messages": [], "return": { - "error_type": "budget_grace_block", - "message": "Refused 'terminal': the iteration budget is exhausted and this is the final grace turn. Only read-only tools may run now. Write your final summary/answer directly instead of calling side-effecting tools." + "error_type": "tool_scope_block", + "message": "'golden_lookup' is not available in this session. Use tool_search to find tools you can call." } } }, - "a70d0db3379feb5e5527653756402b46d6e035feab21dee52fb22b79bf9a362d": { - "input": { - "agent": { - "_in_budget_grace": true - }, - "function_name": "read_file", - "kind": "pre_block", - "name": "budget grace allows readonly", - "tool_scope_block": null - }, - "name": "budget grace allows readonly", - "output": { - "db": [], - "messages": [], - "return": null - } - }, "bea7ae4cb4641e17e86966489c0a9f956843a90af94ac97ce17f3ac39e145189": { "input": { "agent": { diff --git a/tests/golden/tool_gate/runner.py b/tests/golden/tool_gate/runner.py index b70d5b0ff815..7a53656ef886 100644 --- a/tests/golden/tool_gate/runner.py +++ b/tests/golden/tool_gate/runner.py @@ -50,13 +50,6 @@ def resolve_tool_search_unwrap(self, agent, function_name, function_args): return out_name, out_args, block_message, block_result def pre_tool_block_from_builtin_gate(self, agent, function_name, tool_scope_block): - from agent.budget_grace_gate import grace_block_message, is_readonly_grace_tool - - if getattr(agent, "_in_budget_grace", False) and not is_readonly_grace_tool(function_name): - return { - "message": grace_block_message(function_name), - "error_type": "budget_grace_block", - } if tool_scope_block is not None: return { "message": tool_scope_block,