Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions agent/agent_init.py
Original file line number Diff line number Diff line change
Expand Up @@ -1615,6 +1615,21 @@ def _moa_reference_relay(event: str, **kwargs: Any) -> None:
compression_enabled = str(_compression_cfg.get("enabled", True)).lower() in {"true", "1", "yes"}
compression_target_ratio = float(_compression_cfg.get("target_ratio", 0.20))
compression_protect_last = int(_compression_cfg.get("protect_last_n", 20))
# Cap on compression retry rounds before a turn gives up with "max
# compression attempts reached" (compression.max_attempts). Hardcoding 3
# strands sessions that legitimately need more rounds — e.g. a restart
# history reload whose incompressible tool schemas keep the request
# estimate above the threshold even though the messages compress fine
# (the #62605 failure class). Default 3 preserves current behavior, so
# an unset key is behavior-neutral; validated >= 1, hard-capped at 10,
# and a non-integer value falls back to 3.
try:
compression_max_attempts = int(_compression_cfg.get("max_attempts", 3))
except (TypeError, ValueError):
compression_max_attempts = 3
if compression_max_attempts < 1:
compression_max_attempts = 3
compression_max_attempts = min(compression_max_attempts, 10)
# protect_first_n is the number of non-system messages to protect at
# the head, in addition to the system prompt (which is always
# implicitly protected by the compressor). Floor at 0 — a value of
Expand Down Expand Up @@ -1900,6 +1915,7 @@ def _moa_reference_relay(event: str, **kwargs: Any) -> None:
agent.compression_enabled = compression_enabled
agent.compression_in_place = compression_in_place
agent.codex_app_server_auto_compaction = codex_app_server_auto_compaction
agent.max_compression_attempts = compression_max_attempts

# Reject models whose context window is below the minimum required
# for reliable tool-calling workflows (64K tokens).
Expand Down
4 changes: 3 additions & 1 deletion agent/conversation_loop.py
Original file line number Diff line number Diff line change
Expand Up @@ -1125,7 +1125,9 @@ def run_conversation(
retry_count = 0
max_retries = agent._api_max_retries
_retry = TurnRetryState()
max_compression_attempts = 3
# Config-driven via compression.max_attempts (parsed + validated in
# agent_init). Default 3 preserves the prior hardcoded behavior.
max_compression_attempts = getattr(agent, "max_compression_attempts", 3)

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This local is initialized after the pre-API compaction branch, which still uses compression_attempts < 3 and logs /3 on current main (agent/conversation_loop.py:1046-1054). Please resolve the configured cap before that branch and use it there too; otherwise compression.max_attempts: 6 still allows only three preflight compressions on the #62605 path.


finish_reason = "stop"
response = None # Guard against UnboundLocalError if all retries fail
Expand Down
6 changes: 6 additions & 0 deletions cli-config.yaml.example
Original file line number Diff line number Diff line change
Expand Up @@ -429,6 +429,12 @@ compression:
# compression of older turns.
protect_last_n: 20

# Compression retry rounds before a turn gives up with "max compression
# attempts reached" (default: 3, same as the previous hardcoded value).
# Raise (e.g. 6) for tool-schema-heavy sessions where 3 rounds cannot bring
# the request estimate under the threshold. Validated >= 1, hard cap 10.
max_attempts: 3

# Codex app-server (codex CLI runtime) thread-compaction mode. The codex
# agent owns the real thread context on this runtime, so Hermes' summarizer
# cannot shrink it — compaction goes through the app server instead.
Expand Down
5 changes: 5 additions & 0 deletions hermes_cli/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -1425,6 +1425,11 @@ def _ensure_hermes_home_managed(home: Path):
# set this above 0.75 to override the floor.
"target_ratio": 0.20, # fraction of threshold to preserve as recent tail
"protect_last_n": 20, # minimum recent messages to keep uncompressed
"max_attempts": 3, # compression retry rounds before a turn gives up
# with "max compression attempts reached". Raise
# (e.g. 6) for tool-schema-heavy sessions where 3
# rounds cannot clear the request estimate.
# Validated >= 1, hard-capped at 10.
"hygiene_hard_message_limit": 5000, # gateway session-hygiene force-compress threshold by message count
"protect_first_n": 3, # non-system head messages always preserved
# verbatim, in ADDITION to the system prompt
Expand Down
98 changes: 98 additions & 0 deletions tests/agent/test_compression_max_attempts_config.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,98 @@
"""compression.max_attempts — config-driven compression retry cap.

The conversation loop's compression retry cap was hardcoded to 3, stranding
sessions that legitimately need more rounds — e.g. a restart history reload
whose incompressible tool schemas keep the request estimate above the
threshold while the messages themselves compress fine (the #62605 failure
class). The cap is now parsed from ``compression.max_attempts`` in
``agent_init`` and read by the loop via
``getattr(agent, "max_compression_attempts", 3)``.

These tests pin the parse/validate/attach seam: default preserved, custom
value honored, floor and ceiling enforced, garbage tolerated.
"""

from __future__ import annotations

import contextlib
import io
from pathlib import Path

from hermes_state import SessionDB
from run_agent import AIAgent


def _config(max_attempts=None) -> dict:
compression = {
"enabled": True,
"threshold": 0.50,
"target_ratio": 0.20,
"protect_first_n": 3,
"protect_last_n": 20,
}
if max_attempts is not None:
compression["max_attempts"] = max_attempts
return {
"compression": compression,
"prompt_caching": {"cache_ttl": "5m"},
"sessions": {},
"bedrock": {},
}


def _make_agent(monkeypatch, tmp_path: Path, *, max_attempts=None):
from hermes_cli import config as config_mod

monkeypatch.setattr(
config_mod, "load_config", lambda: _config(max_attempts=max_attempts)
)
db = SessionDB(db_path=tmp_path / "state.db")
with contextlib.redirect_stdout(io.StringIO()):
agent = AIAgent(
base_url="https://chatgpt.com/backend-api/codex",
api_key="test-key",
provider="openai-codex",
model="gpt-5.5",
enabled_toolsets=[],
disabled_toolsets=[],
quiet_mode=True,
skip_memory=True,
session_db=db,
session_id="max-attempts-test",
)
return agent


class TestCompressionMaxAttemptsConfig:
def test_default_is_three_when_unset(self, monkeypatch, tmp_path):
agent = _make_agent(monkeypatch, tmp_path)
assert agent.max_compression_attempts == 3

def test_custom_value_is_honored(self, monkeypatch, tmp_path):
agent = _make_agent(monkeypatch, tmp_path, max_attempts=6)
assert agent.max_compression_attempts == 6

def test_hard_capped_at_ten(self, monkeypatch, tmp_path):
agent = _make_agent(monkeypatch, tmp_path, max_attempts=25)
assert agent.max_compression_attempts == 10

def test_zero_and_negative_fall_back_to_default(self, monkeypatch, tmp_path):
agent = _make_agent(monkeypatch, tmp_path, max_attempts=0)
assert agent.max_compression_attempts == 3
agent = _make_agent(monkeypatch, tmp_path, max_attempts=-2)
assert agent.max_compression_attempts == 3

def test_non_integer_falls_back_to_default(self, monkeypatch, tmp_path):
agent = _make_agent(monkeypatch, tmp_path, max_attempts="lots")
assert agent.max_compression_attempts == 3

def test_loop_pickup_degrades_to_default_when_attribute_missing(
self, monkeypatch, tmp_path
):
# The loop reads getattr(agent, "max_compression_attempts", 3): a
# configured agent exposes its value, and an object without the
# attribute (older pickle / minimal stub) degrades to the prior
# hardcoded behavior.
agent = _make_agent(monkeypatch, tmp_path, max_attempts=7)
assert getattr(agent, "max_compression_attempts", 3) == 7
assert getattr(object(), "max_compression_attempts", 3) == 3
Loading