Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 15 additions & 0 deletions agent/agent_init.py
Original file line number Diff line number Diff line change
Expand Up @@ -1327,6 +1327,21 @@ def _apply_agent_section(agent, _agent_cfg):
# "auto" (codex_responses only), true (all api_modes), false, or model substrings.
agent._intent_ack_continuation = _agent_section.get("intent_ack_continuation", "auto")

# Explicit tool-result persistence threshold (tools.tool_result_persist_threshold_chars).
# Results larger than this many chars are persisted to the sandbox and replaced in-context by
# a preview + path, overriding the context-scaled default (100K chars on large models). Invalid
# values (bool, float, non-positive) are ignored rather than clamped to 1, which would persist
# nearly every tool result. ``None`` keeps the context-scaled behavior.
from tools.budget_config import normalize_persist_threshold

_raw_persist_threshold = _cfg_dict(_agent_cfg, "tools").get("tool_result_persist_threshold_chars")
agent._tool_result_persist_threshold_chars = normalize_persist_threshold(_raw_persist_threshold)
if _raw_persist_threshold is not None and agent._tool_result_persist_threshold_chars is None:
_warn_invalid_config_int(
"tools.tool_result_persist_threshold_chars", _raw_persist_threshold,
"must be a positive integer", "the context-scaled default",
)

# Default-on boolean gates: anti-stall guards (notice-only), universal guidance toggles
# (ALL models, unlike enforcement), the local toolchain probe, Bot Mode protocol section.
for _key in (
Expand Down
38 changes: 35 additions & 3 deletions agent/tool_executor.py
Original file line number Diff line number Diff line change
Expand Up @@ -52,7 +52,13 @@
enforce_turn_budget,
extract_persisted_path,
)
from tools.budget_config import BudgetConfig, DEFAULT_BUDGET, budget_for_context_window
from tools.budget_config import (
BudgetConfig,
DEFAULT_BUDGET,
budget_for_context_window,
budget_with_persist_threshold,
normalize_persist_threshold,
)

logger = logging.getLogger(__name__)

Expand Down Expand Up @@ -97,10 +103,36 @@ def _budget_for_agent(agent) -> BudgetConfig:
model switched into mid-session) get a budget proportional to their window so a single large tool result
can't push the request past the model's limit (#23767). Falls back to the default budget when the
context length isn't resolvable.

An explicit ``tools.tool_result_persist_threshold_chars`` (on the agent as
``_tool_result_persist_threshold_chars``) overrides ONLY the per-result threshold; the
small-window turn budget still applies. Values set programmatically by plugins/tests are
validated the same way config values are, so a non-positive or non-int value is ignored
(warned once per agent) rather than clamped to 1. An unresolvable context length does not
discard it -- the explicit cap still applies on top of the unscaled budget.
"""
try:
ctx = getattr(getattr(agent, "context_compressor", None), "context_length", None)
return budget_for_context_window(int(ctx) if ctx else None)
ctx = None
try:
raw_ctx = getattr(getattr(agent, "context_compressor", None), "context_length", None)
ctx = int(raw_ctx) if raw_ctx else None
except Exception:
ctx = None # unresolvable window: unscaled budget, explicit threshold still honored
raw_explicit = getattr(agent, "_tool_result_persist_threshold_chars", None)
explicit = normalize_persist_threshold(raw_explicit)
if raw_explicit is not None and explicit is None:
if not getattr(agent, "_tool_result_persist_threshold_warning_emitted", False):
logger.warning(
"invalid programmatic tools.tool_result_persist_threshold_chars=%r; "
"ignoring (keeping the context-scaled default)", raw_explicit,
)
try:
agent._tool_result_persist_threshold_warning_emitted = True
except Exception:
pass # slot-only test doubles/plugins: the fallback is still safe, only once-per-agent suppression is lost
if explicit is not None:
return budget_with_persist_threshold(explicit, ctx)
return budget_for_context_window(ctx)
except Exception:
return DEFAULT_BUDGET

Expand Down
12 changes: 12 additions & 0 deletions hermes_cli/config_defaults.py
Original file line number Diff line number Diff line change
Expand Up @@ -1837,6 +1837,18 @@ def _aux(timeout, *, reasoning_effort=True, **extra):
# Range 200..60000.
"listing_max_tokens": 4000,
},
# Explicit per-tool-result persistence threshold in characters: results larger than this
# are persisted to the sandbox and replaced in-context by a preview + path
# (tools/tool_result_storage.py).
# None = keep the context-scaled behavior — large models keep the 100K-char cap, small
# models get a budget proportional to their window (tools/budget_config.py).
# A positive int overrides that default. It still sits under the per-tool registry cap
# (web/terminal/x_search register their own 100K max), so a smaller explicit value always
# wins while a larger one is capped by the registry; read_file is always exempt
# (PINNED_THRESHOLDS) to avoid persist->read->persist loops.
# Lowering it (e.g. 20_000) reclaims medium-sized 30-50K-char results from re-sent history
# early; the full originals stay on disk for audit/replay.
"tool_result_persist_threshold_chars": None,
# Remote connector discovery/lifecycle through the Nous tool gateway.
# The flag is the user's off switch; availability additionally requires
# the portal sign-in every managed tool gates on.
Expand Down
138 changes: 138 additions & 0 deletions tests/agent/test_tool_budget_explicit_threshold.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,138 @@
"""Precedence of the explicit tool-result persistence threshold.

``agent.tool_executor._budget_for_agent`` resolves the BudgetConfig used
for tool-result persistence. An explicit
``tools.tool_result_persist_threshold_chars`` value (exposed by agent_init
as ``agent._tool_result_persist_threshold_chars``) must override the
context-scaled default; otherwise context scaling and the DEFAULT_BUDGET
fallback behave exactly as before.
"""

import logging
from types import SimpleNamespace
from unittest.mock import MagicMock

from agent.tool_executor import _budget_for_agent
from tools.budget_config import (
DEFAULT_BUDGET,
DEFAULT_RESULT_SIZE_CHARS,
DEFAULT_TURN_BUDGET_CHARS,
budget_for_context_window,
)


def _agent(persist_threshold=None, context_length=None):
compressor = None
if context_length is not None:
compressor = MagicMock()
compressor.context_length = context_length
return SimpleNamespace(
_tool_result_persist_threshold_chars=persist_threshold,
context_compressor=compressor,
)


class TestExplicitThresholdPrecedence:
def test_explicit_threshold_wins_over_context_scaling(self):
agent = _agent(persist_threshold=20_000, context_length=1_000_000)
cfg = _budget_for_agent(agent)
assert cfg.default_result_size == 20_000
assert cfg.turn_budget == DEFAULT_TURN_BUDGET_CHARS
assert cfg != DEFAULT_BUDGET

def test_explicit_threshold_keeps_small_model_turn_budget(self):
# Regression (#23767): an explicit per-result cap must NOT reset the
# turn budget back to the 200K default on a small-window model.
agent = _agent(persist_threshold=20_000, context_length=65_536)
cfg = _budget_for_agent(agent)
assert cfg.default_result_size == 20_000
scaled = budget_for_context_window(65_536)
assert cfg.turn_budget == scaled.turn_budget
assert cfg.preview_size == scaled.preview_size
assert cfg.turn_budget < DEFAULT_TURN_BUDGET_CHARS

def test_explicit_threshold_ignores_context_failure(self):
agent = _agent(persist_threshold=30_000, context_length="garbage")
cfg = _budget_for_agent(agent)
assert cfg.default_result_size == 30_000

def test_noninteger_explicit_falls_back_to_context_scaling(self):
# A plugin/test setting a non-integer value must not crash resolution.
agent = _agent(persist_threshold="not-an-int", context_length=65_536)
cfg = _budget_for_agent(agent)
assert cfg is not DEFAULT_BUDGET
baseline = _budget_for_agent(
_agent(persist_threshold=None, context_length=65_536)
)
assert cfg.default_result_size == baseline.default_result_size
agent2 = _agent(persist_threshold="not-an-int", context_length=None)
assert _budget_for_agent(agent2) is DEFAULT_BUDGET

def test_no_explicit_uses_context_scaling(self):
agent = _agent(persist_threshold=None, context_length=65_536)
cfg = _budget_for_agent(agent)
assert cfg is not DEFAULT_BUDGET
# 65K model: scaled cap below the 100K default.
assert cfg.default_result_size < DEFAULT_RESULT_SIZE_CHARS

def test_no_explicit_large_context_keeps_default_cap(self):
agent = _agent(persist_threshold=None, context_length=1_000_000)
cfg = _budget_for_agent(agent)
assert cfg.default_result_size == DEFAULT_RESULT_SIZE_CHARS

def test_no_explicit_no_compressor_returns_default_budget(self):
agent = _agent(persist_threshold=None, context_length=None)
assert _budget_for_agent(agent) is DEFAULT_BUDGET

def test_no_explicit_broken_compressor_returns_default_budget(self):
agent = _agent(persist_threshold=None, context_length="garbage")
assert _budget_for_agent(agent) is DEFAULT_BUDGET

def test_zero_explicit_is_treated_as_unset(self):
# agent_init normalizes <=0 to None before it reaches the executor;
# belt-and-braces: a literal 0 must not crash the resolution.
agent = _agent(persist_threshold=0, context_length=65_536)
cfg = _budget_for_agent(agent)
assert cfg.default_result_size < DEFAULT_RESULT_SIZE_CHARS

def test_bool_explicit_falls_back_to_context_scaling(self):
# Regression: int(True) == 1 used to turn a programmatically-set
# boolean into default_result_size == 1 (persist almost everything).
# The executor must normalize through the same single source of truth
# as agent_init and fall back to context scaling.
agent = _agent(persist_threshold=True, context_length=65_536)
cfg = _budget_for_agent(agent)
baseline = _budget_for_agent(
_agent(persist_threshold=None, context_length=65_536)
)
assert cfg.turn_budget == baseline.turn_budget
assert cfg.default_result_size == baseline.default_result_size
assert cfg.default_result_size != 1
assert cfg.turn_budget < DEFAULT_TURN_BUDGET_CHARS
agent2 = _agent(persist_threshold=False, context_length=None)
assert _budget_for_agent(agent2) is DEFAULT_BUDGET

def test_float_explicit_falls_back_to_context_scaling(self):
# int(20000.0) == 20000 is harmless, but int(1.5) == 1 is not; floats
# are rejected wholesale so the rule stays simple and single-sourced.
agent = _agent(persist_threshold=20_000.0, context_length=65_536)
cfg = _budget_for_agent(agent)
baseline = _budget_for_agent(
_agent(persist_threshold=None, context_length=65_536)
)
assert cfg.default_result_size == baseline.default_result_size
assert cfg.default_result_size != 20_000

def test_invalid_programmatic_value_warns_once(self, caplog):
agent = _agent(persist_threshold="not-an-int", context_length=65_536)
with caplog.at_level(logging.WARNING, logger="agent.tool_executor"):
_budget_for_agent(agent)
_budget_for_agent(agent)

matches = [
record
for record in caplog.records
if "invalid programmatic tools.tool_result_persist_threshold_chars"
in record.getMessage()
]
assert len(matches) == 1
139 changes: 139 additions & 0 deletions tests/tools/test_budget_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,8 @@

import dataclasses
import math
from decimal import Decimal
from fractions import Fraction
from unittest.mock import patch

import pytest
Expand All @@ -19,6 +21,8 @@
PINNED_THRESHOLDS,
BudgetConfig,
budget_for_context_window,
budget_with_persist_threshold,
normalize_persist_threshold,
)


Expand Down Expand Up @@ -238,3 +242,138 @@ def test_scaled_small_window_caps_mcp_threshold(self, tmp_path, monkeypatch):
cfg = budget_for_context_window(16_384) # scaled default < 50K
assert cfg.default_result_size < 50_000
assert cfg.resolve_threshold("mcp_tool") == cfg.default_result_size


# ---------------------------------------------------------------------------
# budget_with_persist_threshold() — explicit user-configured cap
# ---------------------------------------------------------------------------


class TestBudgetWithPersistThreshold:
"""Explicit tools.tool_result_persist_threshold_chars path.

Overrides ONLY the per-result cap; turn budget and preview size keep the
context-scaled values (small-model turn-budget protection must survive),
and the pinned/registry guards stay active.
"""

def test_small_model_turn_budget_protection_survives(self):
# 65K-token model: scaled turn budget is ~78,643 chars. An explicit
# per-result cap must NOT reset it back to the 200K default.
cfg = budget_with_persist_threshold(20_000, context_length=65_536)
assert cfg.default_result_size == 20_000
assert cfg.turn_budget == int(65_536 * 4 * 0.30) # 78,643
assert cfg.turn_budget < DEFAULT_TURN_BUDGET_CHARS
assert cfg.preview_size == DEFAULT_PREVIEW_SIZE_CHARS

def test_large_context_turn_budget_caps_at_default(self):
cfg = budget_with_persist_threshold(20_000, context_length=1_000_000)
assert cfg.default_result_size == 20_000
assert cfg.turn_budget == DEFAULT_TURN_BUDGET_CHARS
assert cfg.preview_size == DEFAULT_PREVIEW_SIZE_CHARS

def test_unknown_context_keeps_historical_defaults(self):
cfg = budget_with_persist_threshold(20_000)
assert cfg.default_result_size == 20_000
assert cfg.turn_budget == DEFAULT_TURN_BUDGET_CHARS
assert cfg.preview_size == DEFAULT_PREVIEW_SIZE_CHARS

def test_explicit_threshold_wins_over_default(self):
cfg = budget_with_persist_threshold(20_000)
assert cfg.default_result_size == 20_000
with patch("tools.registry.registry") as mock_registry:
mock_registry.get_max_result_size.return_value = 100_000
assert cfg.resolve_threshold("terminal") == 20_000

def test_smaller_than_registry_cap_wins(self):
cfg = budget_with_persist_threshold(30_000)
with patch("tools.registry.registry") as mock_registry:
# web/terminal/x_search register 100K; explicit 30K must win.
mock_registry.get_max_result_size.return_value = 100_000
assert cfg.resolve_threshold("web_search") == 30_000

def test_larger_than_registry_cap_is_bounded(self):
cfg = budget_with_persist_threshold(200_000)
with patch("tools.registry.registry") as mock_registry:
mock_registry.get_max_result_size.return_value = 100_000
assert cfg.resolve_threshold("terminal") == 100_000

def test_pinned_read_file_stays_exempt(self):
cfg = budget_with_persist_threshold(1)
assert cfg.resolve_threshold("read_file") == float("inf")

def test_nonpositive_input_is_treated_as_unset(self):
# A 0/negative value must NOT clamp to 1 (which would persist almost
# every tool result); it is treated as "not configured" and returns the
# context-scaled budget unchanged.
assert budget_with_persist_threshold(0) is DEFAULT_BUDGET
assert budget_with_persist_threshold(-10) is DEFAULT_BUDGET
assert budget_with_persist_threshold(None) is DEFAULT_BUDGET
scaled = budget_with_persist_threshold(0, context_length=65_536)
assert scaled.turn_budget == int(65_536 * 4 * 0.30)

def test_normalize_rejects_bool(self):
# bool is an int subclass: int(True) == 1 would persist almost every
# tool result. Rejected even when set programmatically (the executor
# path), not just via YAML parsing.
assert normalize_persist_threshold(True) is None
assert normalize_persist_threshold(False) is None

def test_normalize_rejects_float(self):
# int(1.5) == 1 silently truncates into a near-universal persist
# threshold; only ints and whole-number strings are accepted.
assert normalize_persist_threshold(1.5) is None
assert normalize_persist_threshold(20_000.0) is None

def test_normalize_accepts_positive_int_and_whole_str(self):
assert normalize_persist_threshold(20_000) == 20_000
assert normalize_persist_threshold("20000") == 20_000

def test_normalize_rejects_nonpositive_and_garbage(self):
assert normalize_persist_threshold(0) is None
assert normalize_persist_threshold(-1) is None
assert normalize_persist_threshold("abc") is None
assert normalize_persist_threshold([]) is None
assert normalize_persist_threshold("20.5") is None # int() raises

def test_normalize_rejects_decimal_fraction_bytes(self):
# Regression: int(Decimal("1.5")) == 1, int(Fraction(3, 2)) == 1 and
# int(b"20000") all silently coerce -- the same truncation trap as
# float/bool. Only non-bool int and whole-number str are accepted.
assert normalize_persist_threshold(Decimal("1.5")) is None
assert normalize_persist_threshold(Decimal("20000.0")) is None
assert normalize_persist_threshold(Fraction(3, 2)) is None
assert normalize_persist_threshold(Fraction(20000, 1)) is None
assert normalize_persist_threshold(b"20000") is None

def test_normalize_rejects_non_whole_strings(self):
# "20.5" already covered above; exponent/scientific notation and
# empty/whitespace-only strings must not reach int() either.
assert normalize_persist_threshold("1e4") is None
assert normalize_persist_threshold("") is None
assert normalize_persist_threshold(" ") is None
assert normalize_persist_threshold("+20000") == 20_000 # whole, signed
assert normalize_persist_threshold(" 20000 ") == 20_000 # stripped

def test_factory_rejects_decimal_fraction_set_programmatically(self):
scaled = budget_with_persist_threshold(Decimal("1.5"), context_length=65_536)
assert scaled.turn_budget == int(65_536 * 4 * 0.30)
assert scaled.default_result_size == int(65_536 * 4 * 0.15)
scaled = budget_with_persist_threshold(Fraction(3, 2), context_length=65_536)
assert scaled.default_result_size == int(65_536 * 4 * 0.15)

def test_factory_rejects_bool_set_programmatically(self):
# Regression: budget_with_persist_threshold(True) used to yield
# default_result_size == 1 via int(True). Must return the
# context-scaled budget unchanged instead.
assert budget_with_persist_threshold(True) is DEFAULT_BUDGET
assert budget_with_persist_threshold(False) is DEFAULT_BUDGET
scaled = budget_with_persist_threshold(True, context_length=65_536)
assert scaled.turn_budget == int(65_536 * 4 * 0.30)
assert scaled.default_result_size == int(65_536 * 4 * 0.15)
assert scaled.default_result_size != 1

def test_factory_rejects_float_set_programmatically(self):
scaled = budget_with_persist_threshold(20_000.0, context_length=65_536)
assert scaled.turn_budget == int(65_536 * 4 * 0.30)
assert scaled.default_result_size == int(65_536 * 4 * 0.15)
Loading