Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 6 additions & 3 deletions engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -117,7 +117,8 @@ def should_compress_preflight(self, messages):
return rough >= self.threshold_tokens

def compress(self, messages: List[Dict[str, Any]],
current_tokens: int = None) -> List[Dict[str, Any]]:
current_tokens: int = None,
focus_topic: Optional[str] = None) -> List[Dict[str, Any]]:
"""Main compaction entry point.

1. Ingest any new messages into the store
Expand Down Expand Up @@ -169,6 +170,7 @@ def compress(self, messages: List[Dict[str, Any]],
model=self._config.summary_model,
l2_budget_ratio=self._config.l2_budget_ratio,
l3_truncate_tokens=self._config.l3_truncate_tokens,
focus_topic=focus_topic or "",
)

# Step 5: Create DAG node
Expand All @@ -190,7 +192,7 @@ def compress(self, messages: List[Dict[str, Any]],
self._last_compacted_store_id = max(source_store_ids) if source_store_ids else 0

# Step 6: Check if condensation is needed
self._maybe_condense()
self._maybe_condense(focus_topic=focus_topic)

# Step 7: Assemble new active context
compressed = self._assemble_context(messages[0], messages[fresh_tail_start:])
Expand Down Expand Up @@ -382,7 +384,7 @@ def _serialize_messages(self, messages: List[Dict[str, Any]]) -> str:

# -- Internal: condensation --------------------------------------------

def _maybe_condense(self) -> None:
def _maybe_condense(self, focus_topic: Optional[str] = None) -> None:
"""Check if any depth level has enough nodes for condensation."""
max_depth = self._config.incremental_max_depth
if max_depth == 0:
Expand Down Expand Up @@ -418,6 +420,7 @@ def _maybe_condense(self) -> None:
model=self._config.summary_model,
l2_budget_ratio=self._config.l2_budget_ratio,
l3_truncate_tokens=self._config.l3_truncate_tokens,
focus_topic=focus_topic or "",
)

node = SummaryNode(
Expand Down
24 changes: 20 additions & 4 deletions escalation.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@ def _call_llm_for_summary(prompt: str, max_tokens: int,
return None


def _build_l1_prompt(text: str, token_budget: int, depth: int) -> str:
def _build_l1_prompt(text: str, token_budget: int, depth: int, focus_topic: str = "") -> str:
"""Level 1: preserve details."""
depth_guidance = {
0: "Preserve decisions, rationale, constraints, active tasks, file paths, commands, and specific values.",
Expand All @@ -47,22 +47,37 @@ def _build_l1_prompt(text: str, token_budget: int, depth: int) -> str:
}
guidance = depth_guidance.get(depth, depth_guidance[2])

focus_guidance = ""
if focus_topic:
focus_guidance = (
f'Prioritize preserving information related to: "{focus_topic}".\n'
"Spend roughly 60-70% of the summary budget on that topic when relevant.\n"
)

return f"""Summarize this conversation segment for future turns.
{guidance}
Remove repetition and conversational filler.
End with: "Expand for details about: <what was compressed>"
{focus_guidance}

Target ~{token_budget} tokens.

CONTENT:
{text}"""


def _build_l2_prompt(text: str, token_budget: int) -> str:
def _build_l2_prompt(text: str, token_budget: int, focus_topic: str = "") -> str:
"""Level 2: aggressive bullet points."""
focus_guidance = ""
if focus_topic:
focus_guidance = (
f'Prioritize bullets related to: "{focus_topic}" when present.\n'
)

return f"""Compress this into bullet points. Maximum {token_budget} tokens.
Keep only: decisions made, files changed, errors hit, current state.
Drop all reasoning, alternatives considered, and process detail.
{focus_guidance}

CONTENT:
{text}"""
Expand Down Expand Up @@ -97,14 +112,15 @@ def summarize_with_escalation(
model: str = "",
l2_budget_ratio: float = 0.50,
l3_truncate_tokens: int = 512,
focus_topic: str = "",
) -> tuple[str, int]:
"""Run 3-level escalation. Returns (summary, level_used).

Guarantees convergence: level 3 is deterministic and always produces
output shorter than the source.
"""
# Level 1: detailed summary
l1_prompt = _build_l1_prompt(text, token_budget, depth)
l1_prompt = _build_l1_prompt(text, token_budget, depth, focus_topic=focus_topic)
l1_result = _call_llm_for_summary(l1_prompt, token_budget * 2, model=model)

if l1_result and count_tokens(l1_result) < source_tokens:
Expand All @@ -113,7 +129,7 @@ def summarize_with_escalation(

# Level 2: aggressive bullets at reduced budget
l2_budget = int(token_budget * l2_budget_ratio)
l2_prompt = _build_l2_prompt(text, l2_budget)
l2_prompt = _build_l2_prompt(text, l2_budget, focus_topic=focus_topic)
l2_result = _call_llm_for_summary(l2_prompt, l2_budget * 2, model=model)

if l2_result and count_tokens(l2_result) < source_tokens:
Expand Down
21 changes: 21 additions & 0 deletions tests/test_lcm_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,27 @@ def test_get_status(self, engine):
assert "store_messages" in status
assert "dag_nodes" in status

def test_compress_accepts_focus_topic(self, engine, monkeypatch):
import importlib

captured = {}

def mock_summary(**kwargs):
captured["focus_topic"] = kwargs.get("focus_topic")
return "Focused summary.\nExpand for details about: database", 1

lcm_engine_module = importlib.import_module("hermes_lcm.engine")
monkeypatch.setattr(lcm_engine_module, "summarize_with_escalation", mock_summary)

messages = [{"role": "system", "content": "You are a helpful assistant."}]
for i in range(20):
messages.append({"role": "user", "content": f"Question {i}: " + "x" * 200})
messages.append({"role": "assistant", "content": f"Answer {i}: " + "y" * 200})

engine.compress(messages, focus_topic="database schema")

assert captured["focus_topic"] == "database schema"


class TestEngineIngest:
def test_ingest_stores_messages(self, engine):
Expand Down