Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions agent/agent_init.py
Original file line number Diff line number Diff line change
Expand Up @@ -573,6 +573,7 @@ def init_agent(
agent._executing_tools = False
agent._tool_guardrails = ToolCallGuardrailController()
agent._tool_guardrail_halt_decision: ToolGuardrailDecision | None = None
agent._tool_guardrail_halt_count: int = 0

# Interrupt mechanism for breaking out of tool loops
agent._interrupt_requested = False
Expand Down
44 changes: 26 additions & 18 deletions agent/conversation_loop.py
Original file line number Diff line number Diff line change
Expand Up @@ -4744,26 +4744,34 @@ def _perform_api_call(next_api_kwargs):

if agent._tool_guardrail_halt_decision is not None:
decision = agent._tool_guardrail_halt_decision
_turn_exit_reason = "guardrail_halt"
final_response = agent._toolguard_controlled_halt_response(decision)
agent._emit_status(
f"⚠️ Tool guardrail halted {decision.tool_name}: {decision.code}"
f"⚠️ Tool guardrail blocked {decision.tool_name}: {decision.code}"
)
messages.append({"role": "assistant", "content": final_response})
# Emit the halt message to the client so it's not
# indistinguishable from a crash. The stream display
# was flushed (callback(None)) before tool execution,
# but the callback is still alive — fire the text
# through it so SSE/TUI clients see the explanation.
if final_response:
agent._safe_print(f"\n{final_response}\n")
if agent.stream_delta_callback:
try:
agent.stream_delta_callback(final_response)
agent.stream_delta_callback(None)
except Exception:
pass
break

if agent._tool_guardrail_halt_count >= 2:
# Second guardrail halt in the same turn — the model
# already had one rebound chance but failed to
# self-correct. End the turn with a controlled halt.
_turn_exit_reason = "guardrail_halt"
final_response = agent._toolguard_controlled_halt_response(decision)
messages.append({"role": "assistant", "content": final_response})
if final_response:
agent._safe_print(f"\n{final_response}\n")
if agent.stream_delta_callback:
try:
agent.stream_delta_callback(final_response)
agent.stream_delta_callback(None)
except Exception:
pass
break

# First guardrail halt this turn — the synthetic tool
# result is already in messages. Give the model exactly
# one rebound chance to see the error and self-correct.
# Reset the decision so the next before_call (if any)
# can record a second strike.
agent._tool_guardrail_halt_decision = None
continue

# Reset per-turn retry counters after successful tool
# execution so a single truncation doesn't poison the
Expand Down
1 change: 1 addition & 0 deletions agent/turn_context.py
Original file line number Diff line number Diff line change
Expand Up @@ -233,6 +233,7 @@ def build_turn_context(
agent._unicode_sanitization_passes = 0
agent._tool_guardrails.reset_for_turn()
agent._tool_guardrail_halt_decision = None
agent._tool_guardrail_halt_count = 0
_reset_consol = getattr(agent._memory_store, "reset_consolidation_failures", None)
if callable(_reset_consol):
_reset_consol()
Expand Down
13 changes: 10 additions & 3 deletions run_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -5633,9 +5633,16 @@ def _compress_context(self, messages: list, system_message: str, *, approx_token
)

def _set_tool_guardrail_halt(self, decision: ToolGuardrailDecision) -> None:
"""Record the first guardrail decision that should stop this turn."""
if decision.should_halt and self._tool_guardrail_halt_decision is None:
self._tool_guardrail_halt_decision = decision
"""Record guardrail halt decisions with rebound count.

The first halt per turn is a rebound opportunity — the model sees the
synthetic tool result and can self-correct. A second halt in the same
turn triggers the real exit break.
"""
if decision.should_halt:
self._tool_guardrail_halt_count += 1

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This increments once per blocked decision, not once per rebound opportunity. The concurrent executor evaluates every call in a batch before returning; two distinct already-blocked calls can make this 2 before the model has seen either synthetic result, so the loop takes the terminal branch without any rebound. Count a completed guardrail-bearing batch/loop cycle once instead.

if self._tool_guardrail_halt_decision is None:
self._tool_guardrail_halt_decision = decision

def _toolguard_controlled_halt_response(self, decision: ToolGuardrailDecision) -> str:
tool = decision.tool_name or "a tool"
Expand Down
Loading