Skip to content
7 changes: 7 additions & 0 deletions agent/agent_init.py
Original file line number Diff line number Diff line change
Expand Up @@ -2013,6 +2013,12 @@ def _parse_prune_int(raw, default):
compression_in_place = is_truthy_value(
_compression_cfg.get("in_place"), default=True
)
try:
compression_warn_after_compressions = max(
0, int(_compression_cfg.get("warn_after_compressions", 2))
)
except (TypeError, ValueError):
compression_warn_after_compressions = 2
# Opt-in (default False): a micro-compaction pass rewrites already-sent
# history every turn, which breaks the provider prompt-cache prefix on a
# per-turn cadence rather than at an episodic boundary. That is the cost
Expand Down Expand Up @@ -2493,6 +2499,7 @@ def _parse_prune_int(raw, default):
pass
agent.compression_enabled = compression_enabled
agent.compression_in_place = compression_in_place
agent.compression_warn_after_compressions = compression_warn_after_compressions
# Apply micro-compaction settings to the compressor (feature is opt-in)
_cc = getattr(agent, "context_compressor", None)
if _cc is not None and hasattr(_cc, "_micro_compact_enabled"):
Expand Down
3 changes: 2 additions & 1 deletion agent/conversation_compression.py
Original file line number Diff line number Diff line change
Expand Up @@ -3427,7 +3427,8 @@ def _release_lock() -> None:
# storing it on _compression_warning lets replay_compression_warning
# re-deliver it once a late-bound gateway status_callback is wired (#36908).
_cc = agent.context_compressor.compression_count
if _cc >= 2:
_warn_after = int(getattr(agent, "compression_warn_after_compressions", 2) or 0)
if _warn_after > 0 and _cc >= _warn_after:
_cc_msg = (
f"{agent.log_prefix}⚠️ Session compressed {_cc} times — "
f"accuracy may degrade. Consider /new to start fresh."
Expand Down
6 changes: 6 additions & 0 deletions cli-config.yaml.example
Original file line number Diff line number Diff line change
Expand Up @@ -514,6 +514,12 @@ compression:
# head messages, matching the pre-feature behaviour.
protect_first_n: 3

# After this many compactions, Hermes warns that accuracy may degrade and
# suggests /new. Set to 0 to disable the repeated-compression warning entirely.
# Default 2 (warn starting at the 2nd compaction). Takes effect on the next
# agent construction — restart the gateway or start a new session after editing.
warn_after_compressions: 2

# Idle compaction (default: 0 = disabled). When > 0, a session that resumes
# after at least this many seconds of inactivity compacts its accumulated
# history up front, before the first reply, so a long-lived thread you come
Expand Down
1 change: 1 addition & 0 deletions gateway/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -22250,6 +22250,7 @@ async def _run_process_watcher(self, watcher: dict) -> None:
("compression", "codex_app_server_auto"),
("compression", "target_ratio"),
("compression", "protect_last_n"),
("compression", "warn_after_compressions"),
("compression", "proactive_prune_tokens"),
("compression", "proactive_prune_min_result_chars"),
("compression", "proactive_prune_min_reclaim_tokens"),
Expand Down
2 changes: 2 additions & 0 deletions hermes_cli/config_defaults.py
Original file line number Diff line number Diff line change
Expand Up @@ -725,6 +725,8 @@
# session_search and recoverable, not deleted.
# Default True since 2107b86024; set False to
# restore the legacy rotating-compaction path.
"warn_after_compressions": 2, # Show "Session compressed N times…" after this
# many compactions (0 = disable the warning).
"model_thresholds": {}, # Per-model threshold overrides. Keys are
# substring-matched against the model name
# (longest match wins); values replace the
Expand Down
116 changes: 102 additions & 14 deletions tests/agent/test_compression_count_warning_36908.py
Original file line number Diff line number Diff line change
@@ -1,8 +1,8 @@
"""Regression for #36908: the repeated-compression warning must reach the
TUI / gateway, not just CLI stdout.

When a session is compressed >= 2 times, ``compress_context`` warns that
accuracy may degrade. That warning used to go through ``_vprint`` (stdout
When a session is compressed >= warn_after_compressions times, ``compress_context``
warns that accuracy may degrade. That warning used to go through ``_vprint`` (stdout
only), so the Ink TUI / Telegram / Discord never saw it — unlike the two
other compression warnings in the same module, which route through
``_emit_status`` (and store ``_compression_warning`` for late-bound
Expand All @@ -15,10 +15,18 @@
from pathlib import Path
from unittest.mock import MagicMock, patch

import pytest

from hermes_state import SessionDB


def _build_agent_with_db(db: SessionDB, session_id: str, compression_count: int):
def _build_agent_with_db(
db: SessionDB,
session_id: str,
compression_count: int,
*,
warn_after_compressions: int = 2,
):
with patch.dict(os.environ, {"OPENROUTER_API_KEY": "test-key"}):
from run_agent import AIAgent

Expand Down Expand Up @@ -46,22 +54,30 @@ def _build_agent_with_db(db: SessionDB, session_id: str, compression_count: int)
compressor._last_aux_model_failure_model = None
compressor._last_aux_model_failure_error = None
agent.context_compressor = compressor
agent.compression_warn_after_compressions = warn_after_compressions
return agent


def _run_compress(agent) -> list[str]:
emitted: list[str] = []
agent._emit_status = lambda message: emitted.append(message)
messages = [{"role": "user", "content": f"m{i}"} for i in range(20)]
agent._compress_context(messages, "sys", approx_tokens=120_000)
return emitted


def _has_repeated_compression_warning(emitted: list[str]) -> bool:
return any("compressed" in m.lower() and "times" in m.lower() for m in emitted)


def test_repeated_compression_warning_routed_through_emit_status(tmp_path: Path) -> None:
db = SessionDB(db_path=tmp_path / "state.db")
sid = "PARENT_36908"
db.create_session(sid, source="cli")

# compression_count == 2 → the "compressed N times" warning should fire.
agent = _build_agent_with_db(db, sid, compression_count=2)

emitted: list[str] = []
agent._emit_status = lambda message: emitted.append(message)

messages = [{"role": "user", "content": f"m{i}"} for i in range(20)]
agent._compress_context(messages, "sys", approx_tokens=120_000)
emitted = _run_compress(agent)

# The warning reached the gateway-aware channel...
assert any("compressed 2 times" in m.lower() for m in emitted), (
Expand All @@ -78,10 +94,82 @@ def test_no_warning_below_threshold(tmp_path: Path) -> None:

# compression_count == 1 → no repeated-compression warning.
agent = _build_agent_with_db(db, sid, compression_count=1)
emitted: list[str] = []
agent._emit_status = lambda message: emitted.append(message)
emitted = _run_compress(agent)

assert not _has_repeated_compression_warning(emitted)


def test_custom_warn_after_compressions_defers_warning(tmp_path: Path) -> None:
db = SessionDB(db_path=tmp_path / "state.db")
sid = "PARENT_53876_DEFER"
db.create_session(sid, source="cli")

agent = _build_agent_with_db(
db, sid, compression_count=4, warn_after_compressions=5
)
emitted = _run_compress(agent)

assert not _has_repeated_compression_warning(emitted)


def test_custom_warn_after_compressions_fires_at_threshold(tmp_path: Path) -> None:
db = SessionDB(db_path=tmp_path / "state.db")
sid = "PARENT_53876_FIRE"
db.create_session(sid, source="cli")

agent = _build_agent_with_db(
db, sid, compression_count=5, warn_after_compressions=5
)
emitted = _run_compress(agent)

assert any("compressed 5 times" in m.lower() for m in emitted)


def test_warn_after_compressions_zero_disables_warning(tmp_path: Path) -> None:
db = SessionDB(db_path=tmp_path / "state.db")
sid = "PARENT_53876_OFF"
db.create_session(sid, source="cli")

agent = _build_agent_with_db(
db, sid, compression_count=99, warn_after_compressions=0
)
emitted = _run_compress(agent)

assert not _has_repeated_compression_warning(emitted)


def _agent_from_compression_config(warn_after_compressions):
"""Construct AIAgent with load_config returning the given warn threshold.

Exercises the real agent_init coercion path (not setattr).
"""
cfg = {"compression": {"warn_after_compressions": warn_after_compressions}}
with patch.dict(os.environ, {"OPENROUTER_API_KEY": "test-key"}), patch(
"hermes_cli.config.load_config", return_value=cfg
), patch(
"hermes_cli.config.load_config_readonly", return_value=cfg
):
from run_agent import AIAgent

return AIAgent(
api_key="test-key",
base_url="https://openrouter.ai/api/v1",
model="test/model",
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)

messages = [{"role": "user", "content": f"m{i}"} for i in range(20)]
agent._compress_context(messages, "sys", approx_tokens=120_000)

assert not any("compressed" in m.lower() and "times" in m.lower() for m in emitted)
@pytest.mark.parametrize(
"raw, expected",
[
(5, 5),
(0, 0),
(None, 2),
("abc", 2),
],
)
def test_warn_after_compressions_config_to_agent_attribute(raw, expected) -> None:
agent = _agent_from_compression_config(raw)
assert agent.compression_warn_after_compressions == expected
16 changes: 16 additions & 0 deletions tests/gateway/test_agent_cache.py
Original file line number Diff line number Diff line change
Expand Up @@ -90,6 +90,20 @@ def test_compression_threshold_change_busts_cache(self):
assert sig1 != sig2


def test_warn_after_compressions_change_busts_cache(self):
from gateway.run import GatewayRunner

runtime = {"api_key": "k", "base_url": "u", "provider": "p"}
sig1 = GatewayRunner._agent_config_signature(
"m", runtime, [], "",
cache_keys={"compression.warn_after_compressions": 2},
)
sig2 = GatewayRunner._agent_config_signature(
"m", runtime, [], "",
cache_keys={"compression.warn_after_compressions": 5},
)
assert sig1 != sig2

def test_cache_keys_key_order_does_not_matter(self):
"""Signature must be stable regardless of dict key insertion order."""
from gateway.run import GatewayRunner
Expand Down Expand Up @@ -123,6 +137,7 @@ def test_reads_compression_subkeys(self):
"target_ratio": 0.3,
"protect_last_n": 25,
"codex_app_server_auto": "hermes",
"warn_after_compressions": 5,
"some_other_key": "ignored",
}
}
Expand All @@ -133,6 +148,7 @@ def test_reads_compression_subkeys(self):
assert out["compression.target_ratio"] == 0.3
assert out["compression.protect_last_n"] == 25
assert out["compression.codex_app_server_auto"] == "hermes"
assert out["compression.warn_after_compressions"] == 5


def test_missing_keys_yield_none(self):
Expand Down
3 changes: 3 additions & 0 deletions website/docs/user-guide/configuration.md
Original file line number Diff line number Diff line change
Expand Up @@ -792,6 +792,7 @@ compression:
in_place: true # Compact on the same session id (no rotation) — see below
idle_compact_after_seconds: 0 # Opt-in idle compaction (0 = disabled) — see below
hygiene_hard_message_limit: 5000 # Gateway safety valve — see below
warn_after_compressions: 2 # Repeated-compression warning after N compactions (0 = off)
hygiene_timeout_seconds: 30 # Max seconds of NO summary-model output before hygiene compression is cut off
hygiene_total_ceiling_seconds: 600 # Absolute cap on the hygiene wait even while tokens are still streaming
hygiene_failure_cooldown_seconds: 300 # Skip repeated failed hygiene attempts for this session
Expand Down Expand Up @@ -831,6 +832,8 @@ Older configs with `compression.summary_model`, `compression.summary_provider`,

`in_place` (default `true`) controls what happens to the session identity when compaction fires. When `true`, compaction rewrites the message list and rebuilds the system prompt **without rotating the session id** — the conversation keeps one durable id for its whole life (no `parent_session_id` chain, no `name #2` / `#3` renumbering in session lists). Compaction is non-destructive: the live context is compacted, but the pre-compaction turns are soft-archived under the same id (marked inactive/compacted) — still searchable via `session_search` and recoverable, not deleted. Hooks see the mode via the `in_place` field on the `session:compress` event. Set `in_place: false` to restore the legacy behavior where each compaction rotates to a new session id linked to the old one.

`warn_after_compressions` controls when Hermes shows the repeated-compression warning ("Session compressed N times — accuracy may degrade. Consider /new to start fresh."). Default `2` — the warning appears starting at the second compaction and on every subsequent compaction. Set to `0` to disable it entirely, or raise it (e.g. `5`) for long-running sessions where multiple compactions are expected.

`threshold_tokens` sets an optional **absolute token cap** for the compression trigger. When set, compression fires at the lower of the ratio-based `threshold` and this absolute count — so compression never fires later than the user's preferred token number regardless of which model is active. This solves the problem where switching between models with different context windows (e.g. 1M → 400K) shifts the absolute trigger point. The cap is clamped to the model's context length, so setting it higher than the model supports is safe — the ratio-based threshold is used instead. Default `null` (disabled — ratio-based threshold only). The cap survives model switches and fallback activations.

`idle_compact_after_seconds` is an **opt-in, time-based** trigger that complements the size-based `threshold`. Default `0` (disabled). When set above 0, a session that resumes after at least that many seconds of inactivity compacts its accumulated history up front, before the first reply — so a long-lived thread (e.g. a Telegram conversation you come back to hours later) doesn't re-read its full stale context on every subsequent turn. It never fires when the context is already at or below the post-compression target (`threshold × target_ratio`), and it honors the same failure-cooldown, anti-thrash, and per-session lock guards as every automatic compaction. Example: `idle_compact_after_seconds: 1800` compacts after 30 minutes idle.
Expand Down