From 0cb7f376456f011011fc56783dad100eaa2517d5 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 03:27:24 +0700 Subject: [PATCH 01/54] test: pin source observation times in as_of-pinned trajectory fixtures MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fork main CI has been red since the R1 merge: 4 trajectory-stack tests failed in clean CI while appearing to pass in dev environments — test_registered_tool_uses_the_product_compiler_path, test_code_owns_host_envelope_and_selector_proposes_semantics_only, test_named_facet_delta_is_selected_before_the_only_semantic_call, and test_pre_llm_hook_selective_compiler_uses_existing_auxiliary_seam_and_fails_open. Root cause (corrected from the earlier provider-gate hypothesis): these fixtures pin question_date="2026-07-20" but append their source messages WITHOUT a timestamp, so the store records observation at wall-clock "now". reasoning._ground_one enforces the as_of boundary (falling back to ingested_at when observed_at is absent), so once the clock passed 2026-07-20 every grounding call rejected with "source was observed after the question-date boundary" -> operand_grounding_failed -> status "fallback" / state "unknown" / the pre-llm hook failing open without the lcm-selective-evidence block. A time bomb, not a provider gate: with question_date moved past today the same paths compile fine in a clean pytest+numpy venv with no embedding/LLM provider at all. Fix: pin observed_at/timestamp to 2026-07-19T09:00Z (before the pinned question_date) on the appended sources, exactly the idiom the passing tests in the same files already use (e.g. test_exact_span_entity_date_value_unit_role_and_key_are_product_validated). This keeps the real code path exercised — grounding now validates the observation boundary instead of short-circuiting — and stays deterministic forever. No provider fixtures were needed (the selectors and retrieval in these tests are already deterministic fixtures) and no skip markers were added: nothing in these paths requires a provider. Verified in a CI-faithful venv (pytest+numpy only, agent stub, no provider env): the 4 tests pass both with VOYAGE_API_KEY/OPENAI_API_KEY/ ANTHROPIC_API_KEY unset and with dummy values set; full-suite failure set is byte-identical to the pre-fix baseline minus exactly these 4; ruff clean. --- tests/test_evidence_compiler.py | 9 ++++++- tests/test_host_supplied_evidence.py | 37 +++++++++++++++++++++++++--- tests/test_packaging_install.py | 14 +++++++++-- 3 files changed, 53 insertions(+), 7 deletions(-) diff --git a/tests/test_evidence_compiler.py b/tests/test_evidence_compiler.py index 9781b7d07..c8ddbf2c5 100644 --- a/tests/test_evidence_compiler.py +++ b/tests/test_evidence_compiler.py @@ -678,7 +678,14 @@ def retrieve(args): def test_registered_tool_uses_the_product_compiler_path(tmp_path): engine = _engine(tmp_path) - source = _append(engine, "Maya owns the Atlas rollout.") + # Pin the observation time before the pinned question_date: an unpinned + # append is observed "now", which crosses the 2026-07-20 as_of boundary + # once the wall clock passes it and collapses the compile into fallback. + source = _append( + engine, + "Maya owns the Atlas rollout.", + observed_at=datetime(2026, 7, 19, 9, tzinfo=timezone.utc).timestamp(), + ) proposal = _selector( _claim("owner", source, "atlas-owner", entity="Maya", role="user") )({}) diff --git a/tests/test_host_supplied_evidence.py b/tests/test_host_supplied_evidence.py index f4ef734d7..9332bb458 100644 --- a/tests/test_host_supplied_evidence.py +++ b/tests/test_host_supplied_evidence.py @@ -2,6 +2,7 @@ from __future__ import annotations +from datetime import datetime, timezone import json import sys import types @@ -16,6 +17,14 @@ ) from hermes_lcm.store import MessageStore +# Fixtures that pin question_date="2026-07-20" must also pin the observation +# time of their appended sources: an unpinned append is observed "now", which +# crosses the as_of boundary once the wall clock passes the pinned date and +# collapses grounding into compiler_fallback. +_OBSERVED_BEFORE_QUESTION_DATE = datetime( + 2026, 7, 19, 9, tzinfo=timezone.utc +).timestamp() + def _engine(tmp_path): config = LCMConfig(database_path=str(tmp_path / "lcm.db")) @@ -32,10 +41,20 @@ def test_code_owns_host_envelope_and_selector_proposes_semantics_only(tmp_path): owner_text = "Maya owns the Atlas rollout." deadline_text = "The Atlas rollout was due on 2026-07-15." owner_id = engine._store.append( - "session-a", {"role": "user", "content": owner_text} + "session-a", + { + "role": "user", + "content": owner_text, + "timestamp": _OBSERVED_BEFORE_QUESTION_DATE, + }, ) deadline_id = engine._store.append( - "session-a", {"role": "user", "content": deadline_text} + "session-a", + { + "role": "user", + "content": deadline_text, + "timestamp": _OBSERVED_BEFORE_QUESTION_DATE, + }, ) owner = { "exact_ref": f"lcm:{owner_id}:0-{len(owner_text)}", @@ -127,10 +146,20 @@ def test_named_facet_delta_is_selected_before_the_only_semantic_call(tmp_path): baseline_text = "Atlas rollout notes are available." owner_text = "Maya owns the Atlas rollout." baseline_id = engine._store.append( - "session-a", {"role": "user", "content": baseline_text} + "session-a", + { + "role": "user", + "content": baseline_text, + "timestamp": _OBSERVED_BEFORE_QUESTION_DATE, + }, ) owner_id = engine._store.append( - "session-b", {"role": "user", "content": owner_text} + "session-b", + { + "role": "user", + "content": owner_text, + "timestamp": _OBSERVED_BEFORE_QUESTION_DATE, + }, ) baseline = { "exact_ref": f"lcm:{baseline_id}:0-{len(baseline_text)}", diff --git a/tests/test_packaging_install.py b/tests/test_packaging_install.py index 5990f1974..f05455de2 100644 --- a/tests/test_packaging_install.py +++ b/tests/test_packaging_install.py @@ -1,3 +1,4 @@ +from datetime import datetime, timezone from pathlib import Path import importlib import importlib.util @@ -749,8 +750,17 @@ def register_hook(self, name, callback): module.register(ctx) first = "The first of the two purchases cost $20." second = "The second purchase cost $30." - first_id = ctx.engine._store.append("purchase-a", {"role": "user", "content": first}) - second_id = ctx.engine._store.append("purchase-b", {"role": "user", "content": second}) + # Pin the observation time before the pinned question_date ("2026-07-20"): + # an unpinned append is observed "now", which crosses the as_of boundary + # once the wall clock passes the pinned date, so grounding rejects the + # operands and the hook fails open without the selective-evidence block. + observed_at = datetime(2026, 7, 19, 9, tzinfo=timezone.utc).timestamp() + first_id = ctx.engine._store.append( + "purchase-a", {"role": "user", "content": first, "timestamp": observed_at} + ) + second_id = ctx.engine._store.append( + "purchase-b", {"role": "user", "content": second, "timestamp": observed_at} + ) refs = [ {"exact_ref": f"lcm:{first_id}:0-{len(first)}", "quote": first}, {"exact_ref": f"lcm:{second_id}:0-{len(second)}", "quote": second}, From c4052df515d3e767a01db12474b07362e12bc5fd Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 11:18:40 +0700 Subject: [PATCH 02/54] fix: include time-contract sidecars in FTS search() rows MessageStore.search()'s FTS-path SQL still SELECTed the old 12 message columns + rank + snippet, while _row_to_dict() grew to expect 15 message columns (..., ingested_at, observed_at, observed_at_source). The zip() truncation misread the FTS row's rank as ingested_at and snippet as observed_at, and silently dropped observed_at_source. The LIKE fallback path was unaffected. Add the three sidecar columns to the FTS SELECT and derive the rank/snippet column offset from _MESSAGE_SELECT_COLUMNS instead of a hardcoded 12, so a future column addition can't silently desync the two again. Regression: tests/test_runtime_time_contract.py::test_fts_search_preserves_time_contract_fields (fails on 0cb7f37, passes after this fix). Codex finding: stephenschoettler/hermes-lcm#436 discussion_r3641730742 --- store.py | 3 ++- tests/test_runtime_time_contract.py | 24 ++++++++++++++++++++++++ 2 files changed, 26 insertions(+), 1 deletion(-) diff --git a/store.py b/store.py index 3894e625a..483290133 100644 --- a/store.py +++ b/store.py @@ -1165,6 +1165,7 @@ def search(self, query: str, session_id: str | None = None, rows = self._conn.execute( f"""SELECT m.store_id, m.session_id, m.source, m.role, m.content, m.tool_call_id, m.tool_calls, m.tool_name, m.timestamp, m.token_estimate, m.pinned, m.conversation_id, + m.ingested_at, m.observed_at, m.observed_at_source, rank as search_rank, snippet(messages_fts, 0, '>>>', '<<<', '...', 40) as snippet FROM messages_fts fts @@ -1191,7 +1192,7 @@ def search(self, query: str, session_id: str | None = None, raw_primary_values: list[float] = [] for r in rows: d = self._row_to_dict(r) - base_columns = 12 + base_columns = len(_MESSAGE_SELECT_COLUMNS.split(",")) d["search_rank"] = r[base_columns] if len(r) > base_columns else None d["snippet"] = r[base_columns + 1] if len(r) > (base_columns + 1) else "" d["_directness_score"] = _message_directness_score(d.get("role"), d.get("content"), terms, phrases) diff --git a/tests/test_runtime_time_contract.py b/tests/test_runtime_time_contract.py index 94a48d3c7..e9b3fbea5 100644 --- a/tests/test_runtime_time_contract.py +++ b/tests/test_runtime_time_contract.py @@ -103,6 +103,30 @@ def test_legacy_row_migration_backfills_ingest_only(tmp_path): assert row["observed_at_source"] is None +def test_fts_search_preserves_time_contract_fields(tmp_path): + """FTS-path search() rows must carry the same time-sidecar values as a + direct get() -- not the rank/snippet fields misread as ingested_at/ + observed_at once _MESSAGE_SELECT_COLUMNS grew past the FTS query's + hardcoded column count.""" + store = MessageStore(tmp_path / "fts-time.db") + try: + source_time = 1_710_000_000.25 + store_id = store.append( + "session-a", + {"role": "user", "content": "the quick brown fox jumps over the lazy dog", "timestamp": source_time}, + ) + direct = store.get(store_id) + [fts_row] = store.search("quick fox") + finally: + store.close() + + assert fts_row["store_id"] == store_id + assert fts_row["ingested_at"] == direct["ingested_at"] + assert isinstance(fts_row["ingested_at"], float) + assert fts_row["observed_at"] == direct["observed_at"] == source_time + assert fts_row["observed_at_source"] == direct["observed_at_source"] == "host_message_timestamp" + + def test_schema_classifier_accepts_only_declared_time_sidecars(tmp_path): db_path = tmp_path / "classified.db" dag = SummaryDAG(db_path) From fc8ab23d3fa43d7e49ee5d6736302c06f13fb59e Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 11:18:55 +0700 Subject: [PATCH 03/54] fix: release query-view build lease on post-claim publish failure In _persist_compiled_view(), claim_build() could succeed and then publish_ready() could still raise (e.g. a rejected manifest). The generic `except Exception:` handler only recorded query_view_publish_failed in the result payload and never called mark_failed() on the claimed token, so the row stayed status='building' for the full 300s lease. An immediate, correct retry for the same identity got query_view_build_in_progress ("busy") instead of rebuilding. Track the token outside the try block and call store.mark_failed(token, str(exc)) in the except-Exception branch, mirroring the pattern already used in adaptive_retrieval.py's _persist_view(). Regression: tests/test_evidence_compiler.py::test_persist_compiled_view_releases_lease_on_publish_failure (fails on 0cb7f37 -- row stays 'building' and the retry reports busy; passes after this fix). Codex finding: stephenschoettler/hermes-lcm#436 discussion_r3641730749 --- evidence_compiler.py | 5 +++- tests/test_evidence_compiler.py | 53 +++++++++++++++++++++++++++++++++ 2 files changed, 57 insertions(+), 1 deletion(-) diff --git a/evidence_compiler.py b/evidence_compiler.py index f179ea1bd..d7005eb33 100644 --- a/evidence_compiler.py +++ b/evidence_compiler.py @@ -958,6 +958,7 @@ def _persist_compiled_view(result: dict[str, Any], *, engine: Any) -> None: {"status": "unavailable", "reason_code": "query_view_store_unavailable"} ) return + token = None try: identity = _view_identity(request, candidates) dependencies = [ @@ -1010,7 +1011,9 @@ def _persist_compiled_view(result: dict[str, Any], *, engine: Any) -> None: {"status": "busy", "reason_code": "query_view_build_in_progress"} ) return - except Exception: + except Exception as exc: + if token is not None: + store.mark_failed(token, str(exc)) result["persistence"].update( {"status": "error", "reason_code": "query_view_publish_failed"} ) diff --git a/tests/test_evidence_compiler.py b/tests/test_evidence_compiler.py index c8ddbf2c5..c173c0505 100644 --- a/tests/test_evidence_compiler.py +++ b/tests/test_evidence_compiler.py @@ -750,6 +750,59 @@ def test_selective_query_view_persistence_is_same_db_and_default_off(tmp_path): assert count == 1 +def test_persist_compiled_view_releases_lease_on_publish_failure(tmp_path, monkeypatch): + """A claim_build() lease must not survive a post-claim publish_ready() + exception -- otherwise the view sits status='building' for the full + 300s lease and an immediate, correct retry reports busy instead of + rebuilding (F-PR436-2).""" + engine = _engine(tmp_path) + query_views = QueryViewStore(engine._config.database_path) + engine._query_views = query_views + source = _append(engine, "Maya owns the Atlas rollout.") + selector = _selector( + _claim("owner", source, "atlas-owner", entity="Maya", role="user") + ) + try: + monkeypatch.setattr( + QueryViewStore, + "publish_ready", + lambda self, token, **kwargs: (_ for _ in ()).throw( + ValueError("simulated manifest rejection") + ), + ) + failed_result = compile_evidence( + "Who owns the Atlas rollout?", + engine=engine, + baseline_refs=[source], + selector=selector, + enabled=True, + persist_view=True, + ) + assert failed_result["persistence"]["status"] == "error" + assert failed_result["persistence"]["reason_code"] == "query_view_publish_failed" + + row = query_views._conn.execute( + "SELECT status, build_nonce FROM lcm_query_views" + ).fetchone() + assert row["status"] != "building" + assert row["build_nonce"] == "" + + monkeypatch.undo() + retried_result = compile_evidence( + "Who owns the Atlas rollout?", + engine=engine, + baseline_refs=[source], + selector=selector, + enabled=True, + persist_view=True, + ) + finally: + query_views.close() + engine._store.close() + + assert retried_result["persistence"]["status"] == "published" + + def test_selective_persistence_rejects_generic_or_ungrounded_state(tmp_path): engine = _engine(tmp_path) query_views = QueryViewStore(engine._config.database_path) From 7da3a21211ace642bf311faa63d487797656eea9 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 11:19:07 +0700 Subject: [PATCH 04/54] fix: whitelist lcm_query_*/lcm_trajectory_* for schema-stamp repair classify_version_mismatch() treats any extra table not matching a known feature-family prefix as a genuinely-newer-build signature. The prefix list stopped at lcm_assertion, so a profile with an interim schema_version stamp whose only extras are the new query-view (lcm_query_*) or trajectory (lcm_trajectory_*) sidecar tables was misclassified as genuinely_newer instead of an interim stamp -- sending `/lcm doctor repair schema-stamp` down the refusal path even though these are opt-in, marker-gated features that don't bump the core schema. Add "lcm_query" and "lcm_trajectory" to _KNOWN_FEATURE_TABLE_PREFIXES. Both families are new in this PR with no prior deployed shape to drift from, so (like the existing prefix-only families before their own verifiers existed) they intentionally get no early-variant verifier yet -- that stays a separate, future addition if these tables ever need drop-and-rebuild remediation. Regression: tests/test_schema_stamp_remediation.py::test_classify_interim_stamp_with_query_view_and_trajectory_marker_tables (fails on 0cb7f37 -- classifies genuinely_newer; passes after this fix). Codex finding: stephenschoettler/hermes-lcm#436 discussion_r3641730753 --- db_bootstrap.py | 7 +++++-- tests/test_schema_stamp_remediation.py | 26 ++++++++++++++++++++++++++ 2 files changed, 31 insertions(+), 2 deletions(-) diff --git a/db_bootstrap.py b/db_bootstrap.py index 2b2a4563c..048946209 100644 --- a/db_bootstrap.py +++ b/db_bootstrap.py @@ -251,13 +251,16 @@ def refuse_schema_version_too_new(conn: sqlite3.Connection) -> None: _V5_CORE_PRESENCE_ONLY = ("messages_fts", "nodes_fts") # Extra tables are tolerated only when they belong to a known opt-in feature -# family (temporal-rollup / embedding / chunk / assertion) or are FTS5 shadow tables of the -# core FTS indexes. Anything else means a newer build owns the schema. +# family (temporal-rollup / embedding / chunk / assertion / query-view / +# trajectory) or are FTS5 shadow tables of the core FTS indexes. Anything else +# means a newer build owns the schema. _KNOWN_FEATURE_TABLE_PREFIXES = ( "lcm_rollup", "lcm_embedding", "lcm_chunk", "lcm_assertion", + "lcm_query", + "lcm_trajectory", ) # The known opt-in feature families whose derived tables an interim build may diff --git a/tests/test_schema_stamp_remediation.py b/tests/test_schema_stamp_remediation.py index 079443a75..c5f40fdf9 100644 --- a/tests/test_schema_stamp_remediation.py +++ b/tests/test_schema_stamp_remediation.py @@ -177,6 +177,32 @@ def test_classify_interim_stamp_with_feature_marker_tables(tmp_path): conn.close() +def test_classify_interim_stamp_with_query_view_and_trajectory_marker_tables(tmp_path): + """lcm_query_*/lcm_trajectory_* are opt-in, marker-gated sidecars just like + rollup/embedding/chunk/assertion -- an interim stamp plus only these extra + tables must still classify as an interim stamp, not genuinely_newer + (F-PR436-3: the prefix whitelist previously stopped at lcm_assertion).""" + db_path = tmp_path / "lcm.db" + _build_v5_db(db_path) + conn = sqlite3.connect(db_path) + try: + conn.executescript( + """ + CREATE TABLE lcm_query_views (view_id TEXT PRIMARY KEY); + CREATE TABLE lcm_trajectory_corpora (corpus_id TEXT PRIMARY KEY); + """ + ) + conn.commit() + finally: + conn.close() + _stamp(db_path, db_bootstrap.SCHEMA_VERSION + 1) + conn = sqlite3.connect(db_path) + try: + assert classify_version_mismatch(conn) == db_bootstrap.VERSION_MISMATCH_INTERIM_STAMP + finally: + conn.close() + + def test_classify_genuinely_newer_on_unknown_table(tmp_path): db_path = tmp_path / "lcm.db" _build_v5_db(db_path) From 5832cfbb83ec1a317caa3a40053c698999df4c89 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 11:19:34 +0700 Subject: [PATCH 05/54] fix: drop schema-required proposal.operation that the runtime rejects LCM_COMPILE_EVIDENCE's public proposal schema required "operation", but _validate_proposal() rejects any proposal containing an "operation" key (_PROPOSAL_KEYS deliberately excludes it) because operation is a deterministic, code-derived value computed from the question in prepare_evidence_selector() and only ever surfaced to the selector as input (selector_request["operation"]) -- it is never a selector output. A caller who honestly followed the schema's required-field list got selector_schema_invalid on every call; the proposal-mode path was only reachable by silently violating the advertised contract. Remove "operation" from the proposal object's required list and properties -- the schema's additionalProperties:false now correctly rejects it too, matching the runtime exactly. Regression: tests/test_evidence_compiler.py::test_public_proposal_schema_does_not_require_code_derived_operation (fails on 0cb7f37 -- a schema-compliant proposal is rejected with selector_schema_invalid; passes after this fix). Codex finding: stephenschoettler/hermes-lcm#436 discussion_r3641730756 --- schemas.py | 14 ---------- tests/test_evidence_compiler.py | 47 +++++++++++++++++++++++++++++++++ 2 files changed, 47 insertions(+), 14 deletions(-) diff --git a/schemas.py b/schemas.py index 583387056..25c53eefe 100644 --- a/schemas.py +++ b/schemas.py @@ -557,25 +557,11 @@ "additionalProperties": False, "required": [ "version", - "operation", "selections", "missing_facets", ], "properties": { "version": {"type": "string", "enum": ["evidence-selector-v1"]}, - "operation": { - "type": "string", - "enum": [ - "none", - "date_interval", - "date_filter", - "count_distinct", - "sum", - "difference", - "order", - "latest_fact", - ], - }, "requested_facets": { "type": "array", "maxItems": 12, diff --git a/tests/test_evidence_compiler.py b/tests/test_evidence_compiler.py index c173c0505..1b6f668ac 100644 --- a/tests/test_evidence_compiler.py +++ b/tests/test_evidence_compiler.py @@ -17,6 +17,7 @@ derive_evidence_request, ) from hermes_lcm.query_view_store import QueryViewStore +from hermes_lcm.schemas import LCM_COMPILE_EVIDENCE from hermes_lcm.store import MessageStore from hermes_lcm.tools import lcm_compile_evidence @@ -712,6 +713,52 @@ def test_registered_tool_uses_the_product_compiler_path(tmp_path): assert payload["provenance"]["final_prose_cached"] is False +def test_public_proposal_schema_does_not_require_code_derived_operation(tmp_path): + """The proposal schema's required/allowed keys must match what + _validate_proposal actually accepts. operation is a deterministic, + code-derived field surfaced to the selector as input (via + selector_request["operation"]) -- it is never a selector output, so the + public schema must not require (or allow) callers to echo it back + (F-PR436-4: the schema required it while the runtime rejected it).""" + proposal_schema = LCM_COMPILE_EVIDENCE["parameters"]["properties"]["proposal"] + assert "operation" not in proposal_schema["required"] + assert "operation" not in proposal_schema["properties"] + + engine = _engine(tmp_path) + source = _append( + engine, + "Maya owns the Atlas rollout.", + observed_at=datetime(2026, 7, 19, 9, tzinfo=timezone.utc).timestamp(), + ) + # A caller who builds a proposal containing ONLY the schema's required + # keys (the honest, schema-compliant path) must succeed end to end. + schema_compliant_proposal = { + key: value + for key, value in _selector( + _claim("owner", source, "atlas-owner", entity="Maya", role="user") + )({}).items() + if key in proposal_schema["required"] + } + assert set(schema_compliant_proposal) == set(proposal_schema["required"]) + try: + payload = json.loads( + lcm_compile_evidence( + { + "question": "Who owns the Atlas rollout?", + "question_date": "2026-07-20", + "baseline_refs": [source], + "proposal": schema_compliant_proposal, + }, + engine=engine, + ) + ) + finally: + engine._store.close() + + assert payload["status"] == "compiled" + assert payload["reason_code"] != "selector_schema_invalid" + + def test_selective_query_view_persistence_is_same_db_and_default_off(tmp_path): engine = _engine(tmp_path) query_views = QueryViewStore(engine._config.database_path) From 5a2e05310d699e2efbb2207c4f7a3e4323fa3bc1 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 11:19:49 +0700 Subject: [PATCH 06/54] fix: hash requirement description into requirements_digest() requirements_digest() hashed only slot_id and minimum_refs, omitting the requirement description that defines what evidence a slot is supposed to satisfy. Two unrelated questions sharing a QueryViewIdentity (same subject/predicate/scope, plausible for a coarse-grained intent bucket) and the same slot_id/minimum_refs -- but a different description -- produced an identical digest, so start() treated a cached view built for one meaning as a hit for the other and pre-filled the new requirement with unrelated cached evidence. Confirmed end to end: a "Who is the CFO of Acme?" retrieval was served the CEO's cached citation. Fold the (already whitespace-normalized) description into the hashed identity alongside slot_id/minimum_refs. This intentionally narrows one existing warm-reuse case: test_exact_slot_closure_compute_finish_and_warm_reuse previously varied both the literal question text AND the requirement description on its "warm" call and still asserted a hit. That was exercising the exact gap being closed here. Updated it to keep the description constant while still rewording the question, which is what that test is actually meant to demonstrate (warm reuse survives rephrasing, not a description change); the description-varies case now has its own dedicated coverage. Regressions: - tests/test_adaptive_retrieval.py::test_requirements_digest_distinguishes_descriptions - tests/test_adaptive_retrieval.py::test_cached_view_is_not_reused_across_different_requirement_descriptions (both fail on 0cb7f37 -- digests collide and the CFO retrieval reuses the CEO's evidence; pass after this fix). Codex finding: stephenschoettler/hermes-lcm#436 discussion_r3641730764 --- adaptive_retrieval.py | 6 ++- tests/test_adaptive_retrieval.py | 63 +++++++++++++++++++++++++++++++- 2 files changed, 67 insertions(+), 2 deletions(-) diff --git a/adaptive_retrieval.py b/adaptive_retrieval.py index 2ceb8a867..15a7e9d23 100644 --- a/adaptive_retrieval.py +++ b/adaptive_retrieval.py @@ -172,7 +172,11 @@ def public_dict(self, refs: Sequence[str]) -> dict[str, Any]: def requirements_digest(requirements: Sequence[EvidenceRequirement]) -> str: identity = [ - {"slot_id": item.slot_id, "minimum_refs": item.minimum_refs} + { + "slot_id": item.slot_id, + "description": item.description, + "minimum_refs": item.minimum_refs, + } for item in sorted(requirements, key=lambda item: item.slot_id) ] return hashlib.sha256( diff --git a/tests/test_adaptive_retrieval.py b/tests/test_adaptive_retrieval.py index 9d80f9614..07fc0cde1 100644 --- a/tests/test_adaptive_retrieval.py +++ b/tests/test_adaptive_retrieval.py @@ -12,6 +12,8 @@ MAX_CONTEXT_CHARS, MAX_CONTEXT_TOKENS, MAX_RETRIEVAL_ROUNDS, + EvidenceRequirement, + requirements_digest, ) from hermes_lcm.config import LCMConfig from hermes_lcm.engine import LCMEngine @@ -238,12 +240,17 @@ def test_exact_slot_closure_compute_finish_and_warm_reuse(tmp_path): assert "candidate_answer" not in encoded_manifest assert "answer" not in persisted.view["computation_trace"] + # Reword the literal question but keep the requirement description + # identical to the original build: warm reuse must survive rephrasing + # the QUESTION, but requirements_digest() now folds in description + # (F-PR436-5), so varying the description here would correctly bust + # the cache -- that is exercised separately in + # test_cached_view_is_not_reused_across_different_requirement_descriptions. warm = _start( engine, question="How many different cities have I visited?", operation="count_distinct", minimum_refs=2, - description="distinct destinations", ) assert warm["status"] == "ready" assert warm["query_view"]["status"] == "hit" @@ -264,6 +271,60 @@ def test_exact_slot_closure_compute_finish_and_warm_reuse(tmp_path): engine.shutdown() +def test_requirements_digest_distinguishes_descriptions(): + """requirements_digest() must not collide two requirements that share + slot_id/minimum_refs but describe different evidence -- description is + what defines the slot's meaning (F-PR436-5).""" + ceo = EvidenceRequirement.parse( + {"slot_id": "role_holder", "description": "the CEO of Acme", "minimum_refs": 1} + ) + cfo = EvidenceRequirement.parse( + {"slot_id": "role_holder", "description": "the CFO of Acme", "minimum_refs": 1} + ) + assert requirements_digest([ceo]) != requirements_digest([cfo]) + + +def test_cached_view_is_not_reused_across_different_requirement_descriptions(tmp_path): + """Same identity + same slot_id + same minimum_refs, but a DIFFERENT + requirement description, must be a cache miss -- otherwise start() + pre-fills an unrelated question's slot with stale evidence for a + different meaning (F-PR436-5).""" + engine = _engine(tmp_path) + try: + ceo_id = _append(engine, "Maya is the CEO of Acme.") + _append(engine, "Ben is the CFO of Acme.") + + ceo_started = _start( + engine, + question="Who is the CEO of Acme?", + description="the CEO of Acme", + ) + assert ceo_started["query_view"]["status"] == "miss" + found = _search(engine, ceo_started["retrieval_id"], ceo_id) + ceo_citation = found["evidence"][0]["citation"] + finished = _call( + engine, + action="finish", + retrieval_id=ceo_started["retrieval_id"], + resolved_slots=[{"slot_id": "visits", "evidence_refs": [ceo_citation]}], + selected_refs=[ceo_citation], + computation=None, + ) + assert finished["query_view"]["persistence"]["status"] == "published" + + cfo_started = _start( + engine, + question="Who is the CFO of Acme?", + description="the CFO of Acme", + ) + assert cfo_started["query_view"]["status"] == "miss" + assert ceo_citation not in [ + item["citation"] for item in cfo_started.get("evidence", []) + ] + finally: + engine.shutdown() + + def test_corpus_advance_requires_bounded_delta_search(tmp_path): engine = _engine(tmp_path) try: From 046764e506e2ab04f6114bc485a6b21afdab5e57 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Thu, 23 Jul 2026 18:21:27 +0700 Subject: [PATCH 07/54] trajectory: Policy A lexical-floor nucleus slots (#127) Add lexical_floor:int=0 kwarg to TrajectoryStore.query(). When >0, reserve K nucleus slots for the top pure-BM25 states (via _select_with_floor) before filling the rest from the fused order, honouring the same 5/trajectory cap. Default 0 reproduces the historical fused-only selection byte-for-byte (candidate-composition repair for the semantic-magnet SOURCE_MISS bucket). --- trajectory_store.py | 48 ++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 47 insertions(+), 1 deletion(-) diff --git a/trajectory_store.py b/trajectory_store.py index f24d71f17..cb79edf41 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -1584,6 +1584,41 @@ def _select_diverse(rows: Iterable[sqlite3.Row], limit: int) -> list[sqlite3.Row break return selected + @staticmethod + def _select_with_floor( + fused_rows: Sequence[sqlite3.Row], + global_rows: Sequence[sqlite3.Row], + limit: int, + floor_k: int, + ) -> list[sqlite3.Row]: + """Policy A -- reserve ``floor_k`` nucleus slots for the top pure-BM25 + states, then fill the remainder from the fused order. + + The lexical floor guarantees the strongest lexical winners a slot even + when the semantic boost would otherwise let a few semantic-top + trajectories monopolise the nucleus. Both the floor and the fill honour + the same 5-per-trajectory diversity cap as ``_select_diverse``. + """ + selected = list(TrajectoryStore._select_diverse(global_rows, floor_k)) + selected_ids = {int(row["state_id"]) for row in selected} + per_trajectory: dict[str, int] = {} + for row in selected: + trajectory_id = str(row["trajectory_id"]) + per_trajectory[trajectory_id] = per_trajectory.get(trajectory_id, 0) + 1 + for row in fused_rows: + if len(selected) >= limit: + break + state_id = int(row["state_id"]) + if state_id in selected_ids: + continue + trajectory_id = str(row["trajectory_id"]) + if per_trajectory.get(trajectory_id, 0) >= 5: + continue + selected.append(row) + selected_ids.add(state_id) + per_trajectory[trajectory_id] = per_trajectory.get(trajectory_id, 0) + 1 + return selected[:limit] + def _fts_rows( self, expression: str, @@ -1624,12 +1659,14 @@ def query( image_limit: int = 8, include_adjacent: bool = True, text_char_limit: int = 2_000, + lexical_floor: int = 0, ) -> tuple[TrajectoryHit, ...]: if self.status != "complete": raise CorpusIdentityError("trajectory corpus must be finalized before query") candidate_limit = min(max(1, int(candidate_limit)), _MAX_CANDIDATES) limit = min(max(1, int(limit)), _MAX_RESULTS) image_limit = min(max(0, int(image_limit)), _MAX_IMAGES) + lexical_floor = min(max(0, int(lexical_floor)), _MAX_RESULTS) text_char_limit = min( max(256, int(text_char_limit)), _MAX_QUERY_TEXT_CHARS, @@ -1734,7 +1771,16 @@ def query( adjacent_reserve = min(6, limit // 3) if include_adjacent else 0 nucleus_limit = max(1, limit - adjacent_reserve) - selected = self._select_diverse(rows, nucleus_limit) + if lexical_floor > 0: + # Policy A (candidate-composition repair, issue #127): guarantee the + # top pure-BM25 states a nucleus slot before the fused order fills + # the rest. ``lexical_floor == 0`` (default) is byte-identical to the + # historical fused-only selection below. + selected = self._select_with_floor( + rows, global_rows, nucleus_limit, lexical_floor + ) + else: + selected = self._select_diverse(rows, nucleus_limit) selected_ids = {int(row["state_id"]) for row in selected} match_kind_by_id = { int(row["state_id"]): candidate_kind.get(int(row["state_id"]), "fts") From 1f0078094e69324432120b4c50e31a42b9232d0d Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Thu, 23 Jul 2026 18:22:05 +0700 Subject: [PATCH 08/54] trajectory: Policy D per-arm quota union (#127) Add arm_quota:tuple|None=None kwarg to TrajectoryStore.query(). When set, _merge_arms round-robins a pure-BM25 arm and the semantic/fused arm into the nucleus by a q_lex:q_sem quota (dedup + backfill, arm order as tie-break), strictly generalising Policy A. Default None reproduces the historical selection byte-for-byte. --- trajectory_store.py | 58 ++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 57 insertions(+), 1 deletion(-) diff --git a/trajectory_store.py b/trajectory_store.py index cb79edf41..cb6462fe0 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -1619,6 +1619,49 @@ def _select_with_floor( per_trajectory[trajectory_id] = per_trajectory.get(trajectory_id, 0) + 1 return selected[:limit] + @staticmethod + def _merge_arms( + arm_lex: Sequence[sqlite3.Row], + arm_sem: Sequence[sqlite3.Row], + limit: int, + q_lex: int, + q_sem: int, + ) -> list[sqlite3.Row]: + """Policy D -- round-robin a pure-lexical arm and the semantic/fused arm + into the nucleus by a ``q_lex:q_sem`` quota. + + Strictly generalises Policy A: the lexical arm is guaranteed its quota + (serving the SOURCE_MISS bucket) while the semantic arm keeps its own + quota (preserving the semantic gains). Deduped by ``state_id``; a short + arm is backfilled by the other (a skipped duplicate does not consume a + quota slot). Arm order is the deterministic tie-break. + """ + selected: list[sqlite3.Row] = [] + seen: set[int] = set() + + def _pull(arm: Sequence[sqlite3.Row], start: int, quota: int) -> int: + added = 0 + index = start + while index < len(arm) and added < quota and len(selected) < limit: + row = arm[index] + index += 1 + state_id = int(row["state_id"]) + if state_id in seen: + continue + selected.append(row) + seen.add(state_id) + added += 1 + return index + + lex_i = sem_i = 0 + while len(selected) < limit and (lex_i < len(arm_lex) or sem_i < len(arm_sem)): + next_lex = _pull(arm_lex, lex_i, q_lex) + next_sem = _pull(arm_sem, sem_i, q_sem) + if next_lex == lex_i and next_sem == sem_i: + break + lex_i, sem_i = next_lex, next_sem + return selected[:limit] + def _fts_rows( self, expression: str, @@ -1660,6 +1703,7 @@ def query( include_adjacent: bool = True, text_char_limit: int = 2_000, lexical_floor: int = 0, + arm_quota: tuple[int, int] | None = None, ) -> tuple[TrajectoryHit, ...]: if self.status != "complete": raise CorpusIdentityError("trajectory corpus must be finalized before query") @@ -1771,7 +1815,19 @@ def query( adjacent_reserve = min(6, limit // 3) if include_adjacent else 0 nucleus_limit = max(1, limit - adjacent_reserve) - if lexical_floor > 0: + if arm_quota is not None: + # Policy D (candidate-composition repair, issue #127): round-robin a + # pure-lexical arm and the semantic/fused arm into the nucleus by the + # requested quota. Superset of Policy A; ``arm_quota is None`` + # (default) is byte-identical to the historical selection below. + q_lex = max(0, int(arm_quota[0])) + q_sem = max(0, int(arm_quota[1])) + arm_lex = self._select_diverse(global_rows, nucleus_limit) + arm_sem = self._select_diverse(rows, nucleus_limit) + selected = self._merge_arms( + arm_lex, arm_sem, nucleus_limit, q_lex, q_sem + ) + elif lexical_floor > 0: # Policy A (candidate-composition repair, issue #127): guarantee the # top pure-BM25 states a nucleus slot before the fused order fills # the rest. ``lexical_floor == 0`` (default) is byte-identical to the From 4a3ce7b656d0b0f5d74e7c290dc677c8adb0662f Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Thu, 23 Jul 2026 18:33:38 +0700 Subject: [PATCH 09/54] bench: provider-free H3.1 composition replay harness (#127) Component instrument for the candidate-composition repair. Injects the frozen H1.3 semantic source ranks into TrajectoryStore.query() (no provider/model call), replays from read-only frozen DB copies, and: - GOLDEN GATE: defaults reproduce recorded delivered_evidence_refs byte-for-byte; - RECOVERY: vanished genuine-loss refs re-admitted (by h2 bucket); - CEILING: vanished refs absent from global_rows top-128 (un-recoverable at seam); - PRESERVATION: stable-correct delivered refs dropped (ref-level + source-level); - latency p95 over a 50-query replay. Emits the full A/K + D/quota sweep table + JSON; does not pick a shipping knob. --- benchmarking/h3_composition_replay.py | 446 ++++++++++++++++++++++++++ 1 file changed, 446 insertions(+) create mode 100644 benchmarking/h3_composition_replay.py diff --git a/benchmarking/h3_composition_replay.py b/benchmarking/h3_composition_replay.py new file mode 100644 index 000000000..e71853749 --- /dev/null +++ b/benchmarking/h3_composition_replay.py @@ -0,0 +1,446 @@ +#!/usr/bin/env python3 +"""Provider-free replay harness for the H3.1 candidate-composition repair (#127). + +This is the *component instrument* for the two composition policies added to +``TrajectoryStore.query()`` (Policy A ``lexical_floor`` and Policy D +``arm_quota``). It never calls an embedding provider or any model/API: the +semantic source ranks are INJECTED from the frozen H1.3 telemetry (the recorded +``source_candidate_ranks``), and every other input is recomputed deterministically +from a read-only COPY of the frozen corpus DB. + +Pipeline (mirrors ``TrajectoryStore.query`` exactly): + 1. For each qid, take the verbatim question text (the string the reader harness + passed to ``query()``), open the frozen DB copy read-only, and inject the + recorded 12 semantic source ranks in place of ``_semantic_source_ranks``. + 2. GOLDEN GATE: run the shipped policy (defaults) and assert it reproduces the + recorded ``delivered_evidence_refs`` byte-for-byte. + 3. Swap in a policy knob and re-deliver; measure per knob: + RECOVERY -- vanished loss-target refs (genuine-loss ground truth) + re-admitted to the delivered top-16, split by bucket; + PRESERVATION -- stable-correct questions whose recorded delivered refs + drop out (must be <= 2); + CEILING -- vanished refs absent from ``global_rows`` top-128 + (un-recoverable at this seam -> a recall problem). + 4. Latency: p95 over a 50-query replay (default vs knob). + +Honesty boundary: the reader/judge is NOT replayed, so this proves evidence +DELIVERY recovery + preservation (the necessary condition), not the correctness +flip. The end-to-end +correct delta needs a provider-gated full benchmark re-run. +""" +from __future__ import annotations + +import argparse +import importlib.util +import json +import re +import sqlite3 +import statistics +import sys +import time +from pathlib import Path +from typing import Any + +# --- defaults (frozen H1.3 assets; override on the CLI) ---------------------- +_REPO_ROOT = Path(__file__).resolve().parent.parent +_H1 = Path( + "/Volumes/LEXAR/Codex/session-notes/2026-07-23/hermes-benchprog-h1/artifacts" +) +_H31 = Path( + "/Volumes/LEXAR/Codex/session-notes/2026-07-23/hermes-benchprog-h1-h3.1/artifacts" +) +_RUN_ROOT = Path( + "/Volumes/LEXAR/Codex/benchmarks/longmemeval-v2/runs/bench-h1-v2clean-2026-07-23" +) +_REF_RE = re.compile(r"^trajectory://[^/]+/(?P[^/]+)/state/(?P\d+)$") + + +def _bootstrap_package(repo_root: Path) -> Any: + """Register the plugin dir as the ``hermes_lcm`` package (mirrors conftest).""" + pkg = "hermes_lcm" + if pkg in sys.modules: + return sys.modules[pkg] + parent = str(repo_root.parent) + if parent not in sys.path: + sys.path.insert(0, parent) + spec = importlib.util.spec_from_file_location( + pkg, str(repo_root / "__init__.py"), + submodule_search_locations=[str(repo_root)], + ) + mod = importlib.util.module_from_spec(spec) + mod.__path__ = [str(repo_root)] + mod.__package__ = pkg + sys.modules[pkg] = mod + for py_file in repo_root.glob("*.py"): + if py_file.name == "__init__.py": + continue + sub_name = f"{pkg}.{py_file.stem}" + if sub_name in sys.modules: + continue + sub_spec = importlib.util.spec_from_file_location( + sub_name, str(py_file), submodule_search_locations=[] + ) + sub_mod = importlib.util.module_from_spec(sub_spec) + sub_mod.__package__ = pkg + sys.modules[sub_name] = sub_mod + setattr(mod, py_file.stem, sub_mod) + try: + sub_spec.loader.exec_module(sub_mod) + except Exception: + pass # unrelated modules (engine needs `agent`) may fail; ignore + return mod + + +def _question_text(question_field: Any) -> str: + """The exact query string the reader harness passes to ``query()``. + + Mirrors ``get_question_components``: a plain string is used verbatim; a + multimodal ``{text, image}`` question contributes only its text (the image + is input-only and never reaches the trajectory store). + """ + if isinstance(question_field, str): + return question_field + return str(question_field["text"]) + + +def _qids(labelled: list[str]) -> set[str]: + """`domain/qid` -> `qid`.""" + return {item.split("/", 1)[1] for item in labelled} + + +def _percentile(values: list[float], pct: float) -> float: + if not values: + return 0.0 + ordered = sorted(values) + rank = max(0, min(len(ordered) - 1, int(round((pct / 100.0) * (len(ordered) - 1))))) + return ordered[rank] + + +class ReplayContext: + def __init__(self, run_root: Path, h1: Path, h31: Path) -> None: + pkg = _bootstrap_package(_REPO_ROOT) + self._ts = sys.modules["hermes_lcm.trajectory_store"] + self.h1 = h1 + self.h31 = h31 + self.questions: dict[str, tuple[str, str]] = {} + for domain in ("web", "enterprise"): + for item in json.loads( + (run_root / "runtime_inputs" / domain / "questions.json").read_text() + ): + self.questions[item["id"]] = (domain, _question_text(item["question"])) + self.stores = { + domain: self._open_store(h31 / "db-copies" / f"{domain}.lcm.db", domain) + for domain in ("web", "enterprise") + } + self._state_id_cache: dict[tuple[str, str, int], int | None] = {} + del pkg + + def _open_store(self, db_path: Path, domain: str): + conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True) + identity_json = json.loads( + conn.execute( + "SELECT identity_json FROM lcm_trajectory_corpora WHERE singleton=1" + ).fetchone()[0] + ) + conn.close() + identity = self._ts.CorpusIdentity( + dataset_name=identity_json["dataset_name"], + dataset_revision=identity_json["dataset_revision"], + harness_commit=identity_json["harness_commit"], + tier=identity_json["tier"], + domain=identity_json["domain"], + ingest_config_digest=identity_json.get("ingest_config_digest", ""), + ) + base = self._ts.TrajectoryStore + + class _ReplayStore(base): # type: ignore[misc, valid-type] + injected: list[tuple[int, float]] = [] + + def _semantic_source_ranks(self, query: str): # noqa: ARG002 + return list(self.injected) + + return _ReplayStore( + db_path, identity, asset_root=db_path.parent, + read_only=True, semantic_top_trajectories=12, + ) + + def trace(self, qid: str) -> dict[str, Any]: + domain, _ = self.questions[qid] + path = ( + self.h31 / "query_traces" / domain / "query_traces" / qid + / "hermes_lcm_semantic_telemetry.json" + ) + return json.loads(path.read_text()) + + def _injected_ranks(self, qid: str) -> list[tuple[int, float]]: + ranks = sorted(self.trace(qid)["source_candidate_ranks"], key=lambda r: r["rank"]) + return [(int(r["source_id"]), float(r["score"])) for r in ranks] + + def deliver(self, qid: str, **kwargs: Any) -> list[str]: + """Delivered top-16 exact-refs for ``qid`` under the given policy kwargs.""" + domain, text = self.questions[qid] + store = self.stores[domain] + store.injected = self._injected_ranks(qid) + store.query( + text, candidate_limit=128, limit=16, image_limit=0, + include_adjacent=True, text_char_limit=2000, **kwargs, + ) + return list(store.last_query_telemetry()["delivered_evidence_refs"]) + + def global_top_state_ids(self, qid: str, limit: int = 128) -> set[int]: + domain, text = self.questions[qid] + store = self.stores[domain] + expression = store._fts_expression(text) + if not expression: + return set() + rows = store._fts_rows(expression, limit) + return {int(row["state_id"]) for row in rows} + + def ref_to_state_id(self, qid: str, ref: str) -> int | None: + match = _REF_RE.match(ref) + if not match: + return None + from urllib.parse import unquote + + traj = unquote(match.group("traj")) + state_index = int(match.group("state")) + domain, _ = self.questions[qid] + key = (domain, traj, state_index) + if key in self._state_id_cache: + return self._state_id_cache[key] + store = self.stores[domain] + row = store._conn.execute( + """ + SELECT s.state_id FROM lcm_trajectory_states s + JOIN lcm_trajectory_sources src ON src.source_id = s.source_id + WHERE src.trajectory_id = ? AND s.state_index = ? + """, + (traj, state_index), + ).fetchone() + value = int(row[0]) if row is not None else None + self._state_id_cache[key] = value + return value + + +def golden_gate(ctx: ReplayContext) -> dict[str, Any]: + passed = 0 + failures: list[str] = [] + qids = sorted(ctx.questions) + for qid in qids: + recorded = ctx.trace(qid)["delivered_evidence_refs"] + got = ctx.deliver(qid) + if got == recorded: + passed += 1 + else: + failures.append(qid) + return {"total": len(qids), "passed": passed, "failures": failures} + + +def load_ground_truth(ctx: ReplayContext) -> dict[str, Any]: + flip = json.loads((ctx.h1 / "old-rescore" / "flip_reconciliation_32_47.json").read_text()) + recon = json.loads((ctx.h1 / "old-rescore" / "reconciliation_sets.json").read_text()) + lost_analysis = { + e["qid"]: e + for e in json.loads((ctx.h1 / "H1P3-preservation-scratch" / "lost_analysis.json").read_text()) + } + h2 = { + e["qid"]: e.get("cause_norm") or e.get("cause") + for e in json.loads((ctx.h1 / "h2-attribution-full.json").read_text()) + } + genuine = _qids(flip["of_32_still_loss_under_rescore"]) + preserved = _qids(recon["preserved"]) + # recoverable universe = genuine losses that actually changed evidence + recovery_targets: dict[str, dict[str, Any]] = {} + for qid in genuine: + vanished = lost_analysis.get(qid, {}).get("vanished", []) + recovery_targets[qid] = { + "vanished": list(vanished), + "bucket": h2.get(qid, "(no-h2)"), + } + return { + "genuine_loss_qids": sorted(genuine), + "preserved_qids": sorted(preserved), + "recovery_targets": recovery_targets, + } + + +def measure_ceiling(ctx: ReplayContext, recovery_targets: dict[str, dict[str, Any]]) -> dict[str, Any]: + """Per vanished ref: is its state present in global_rows top-128?""" + per_ref: dict[str, dict[str, bool]] = {} + for qid, info in recovery_targets.items(): + if not info["vanished"]: + continue + top = ctx.global_top_state_ids(qid, 128) + for ref in info["vanished"]: + state_id = ctx.ref_to_state_id(qid, ref) + per_ref.setdefault(qid, {})[ref] = state_id is not None and state_id in top + total = sum(len(v) for v in per_ref.values()) + recoverable = sum(1 for refs in per_ref.values() for present in refs.values() if present) + return { + "per_ref_in_global_top128": per_ref, + "total_vanished_refs": total, + "recoverable": recoverable, + "absent_ceiling": total - recoverable, + } + + +def evaluate_knob( + ctx: ReplayContext, + ground: dict[str, Any], + ceiling: dict[str, Any], + knob_kwargs: dict[str, Any], +) -> dict[str, Any]: + targets = ground["recovery_targets"] + per_ref = ceiling["per_ref_in_global_top128"] + # RECOVERY (recoverable refs re-admitted), split by bucket + readmit = 0 + by_bucket: dict[str, dict[str, int]] = {} + for qid, info in targets.items(): + if not info["vanished"]: + continue + delivered = set(ctx.deliver(qid, **knob_kwargs)) + bucket = info["bucket"] + slot = by_bucket.setdefault(bucket, {"recoverable": 0, "readmitted": 0}) + for ref in info["vanished"]: + if not per_ref.get(qid, {}).get(ref, False): + continue # un-recoverable at this seam; excluded from the pool + slot["recoverable"] += 1 + if ref in delivered: + slot["readmitted"] += 1 + readmit += 1 + # PRESERVATION (stable-correct delivered refs dropped). Two lenses: + # ref-level -- ANY recorded delivered ref drops out (the frozen gate); + # source-level -- a whole delivered TRAJECTORY vanishes (states merely + # reshuffled within a still-present trajectory are far less + # likely to change correctness than an entirely lost source). + def _sources(refs: set[str]) -> set[str]: + out = set() + for ref in refs: + match = _REF_RE.match(ref) + if match: + out.add(match.group("traj")) + return out + + disturbed: list[str] = [] + source_disturbed: list[str] = [] + total_dropped = 0 + for qid in ground["preserved_qids"]: + recorded = set(ctx.trace(qid)["delivered_evidence_refs"]) + delivered = set(ctx.deliver(qid, **knob_kwargs)) + dropped = recorded - delivered + if dropped: + disturbed.append(qid) + total_dropped += len(dropped) + if _sources(recorded) - _sources(delivered): + source_disturbed.append(qid) + recoverable = ceiling["recoverable"] + return { + "knob": knob_kwargs, + "recovery_readmitted": readmit, + "recovery_recoverable": recoverable, + "recovery_pct": (readmit / recoverable * 100.0) if recoverable else 0.0, + "recovery_by_bucket": by_bucket, + "preservation_disturbed_count": len(disturbed), + "preservation_disturbed_qids": disturbed, + "preservation_total_refs_dropped": total_dropped, + "preservation_source_disturbed_count": len(source_disturbed), + "preservation_source_disturbed_qids": source_disturbed, + } + + +def measure_latency(ctx: ReplayContext, sample_qids: list[str], knob_kwargs: dict[str, Any]) -> dict[str, float]: + latencies: list[float] = [] + for qid in sample_qids: + start = time.perf_counter() + ctx.deliver(qid, **knob_kwargs) + latencies.append((time.perf_counter() - start) * 1000.0) + return { + "p50_ms": _percentile(latencies, 50), + "p95_ms": _percentile(latencies, 95), + "mean_ms": statistics.fmean(latencies), + } + + +def _knob_label(kwargs: dict[str, Any]) -> str: + if "lexical_floor" in kwargs: + return f"A(K={kwargs['lexical_floor']})" + if "arm_quota" in kwargs: + q = kwargs["arm_quota"] + return f"D(q={q[0]},{q[1]})" + return "default" + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--run-root", type=Path, default=_RUN_ROOT) + parser.add_argument("--h1-artifacts", type=Path, default=_H1) + parser.add_argument("--h31-artifacts", type=Path, default=_H31) + parser.add_argument("--out", type=Path, default=_H31 / "h3-replay-sweep.json") + parser.add_argument("--latency-sample", type=int, default=50) + args = parser.parse_args() + + ctx = ReplayContext(args.run_root, args.h1_artifacts, args.h31_artifacts) + + print("== GOLDEN GATE (defaults reproduce recorded delivery) ==", flush=True) + golden = golden_gate(ctx) + print(f" {golden['passed']}/{golden['total']} byte-identical; " + f"failures={golden['failures'][:10]}", flush=True) + + ground = load_ground_truth(ctx) + ceiling = measure_ceiling(ctx, ground["recovery_targets"]) + print("\n== CEILING (recoverability of vanished loss refs at this seam) ==") + print(f" vanished refs (genuine losses): {ceiling['total_vanished_refs']}") + print(f" recoverable (in global top-128): {ceiling['recoverable']}") + print(f" absent / un-recoverable ceiling: {ceiling['absent_ceiling']}") + + knobs: list[dict[str, Any]] = [{"lexical_floor": k} for k in (1, 2, 3, 4)] + knobs += [{"arm_quota": q} for q in ((6, 5), (8, 3), (5, 6))] + + sample = sorted(ctx.questions)[: args.latency_sample] + baseline_latency = measure_latency(ctx, sample, {}) + + results = [] + for knob in knobs: + row = evaluate_knob(ctx, ground, ceiling, knob) + row["latency"] = measure_latency(ctx, sample, knob) + row["latency_pct_vs_default"] = ( + (row["latency"]["p95_ms"] / baseline_latency["p95_ms"] - 1.0) * 100.0 + if baseline_latency["p95_ms"] + else 0.0 + ) + row["label"] = _knob_label(knob) + results.append(row) + + print("\n== SWEEP TABLE (component gate is FROZEN; shipping knob NOT picked here) ==") + header = ( + f"{'knob':<11}{'recov%':>8}{'readmit':>9}{'recoverbl':>10}" + f"{'refDist/154':>13}{'srcDist/154':>13}{'refsDropd':>10}{'p95 Δ%':>9}" + ) + print(header) + print("-" * len(header)) + for row in results: + print( + f"{row['label']:<11}{row['recovery_pct']:>7.1f}%" + f"{row['recovery_readmitted']:>9}{row['recovery_recoverable']:>10}" + f"{row['preservation_disturbed_count']:>13}" + f"{row['preservation_source_disturbed_count']:>13}" + f"{row['preservation_total_refs_dropped']:>10}" + f"{row['latency_pct_vs_default']:>+8.1f}%" + ) + + payload = { + "golden_gate": golden, + "ceiling": {k: v for k, v in ceiling.items() if k != "per_ref_in_global_top128"}, + "ground_truth_sizes": { + "genuine_loss_qids": len(ground["genuine_loss_qids"]), + "preserved_qids": len(ground["preserved_qids"]), + }, + "baseline_latency": baseline_latency, + "sweep": results, + } + args.out.write_text(json.dumps(payload, indent=2)) + print(f"\nwrote {args.out}") + return 0 if golden["passed"] == golden["total"] else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) From bbbeb4559bcf1eecd2e2c82b6f85cb6856d22c83 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Thu, 23 Jul 2026 18:33:38 +0700 Subject: [PATCH 10/54] test: candidate-composition policies A + D on a synthetic magnet (#127) Reproduce the semantic-magnet SOURCE_MISS displacement synthetically and assert: default byte-compat (magnet displacement preserved), Policy A/D re-admit the displaced lexical winner, _merge_arms dedup+backfill order, and both policies no-op without semantic ranks. --- tests/test_trajectory_composition_policies.py | 198 ++++++++++++++++++ 1 file changed, 198 insertions(+) create mode 100644 tests/test_trajectory_composition_policies.py diff --git a/tests/test_trajectory_composition_policies.py b/tests/test_trajectory_composition_policies.py new file mode 100644 index 000000000..3eff31f88 --- /dev/null +++ b/tests/test_trajectory_composition_policies.py @@ -0,0 +1,198 @@ +"""Candidate-composition repair policies (issue #127): lexical floor + arm quota. + +These exercise the semantic-"magnet" pathology on a synthetic corpus: a cluster +of semantic-top distractor trajectories monopolises the fused nucleus and +displaces the strongest pure-lexical winner (a SOURCE_MISS). Policy A +(``lexical_floor``) and Policy D (``arm_quota``) must re-admit the lexical +winner, while the defaults must reproduce the displaced (pre-repair) delivery +byte-for-byte. +""" + +from __future__ import annotations + +import hashlib +from pathlib import Path + +from hermes_lcm.trajectory_store import ( + CorpusIdentity, + TrajectorySource, + TrajectoryState, + TrajectoryStore, +) + + +class MagnetProvider: + """Ranks any trajectory whose text mentions 'profile toolbar' as semantically + top, regardless of the query -- the coarse whole-trajectory 'magnet'.""" + + provider_id = "fake" + model_id = "fake-trajectory-v1" + dim = 2 + + def __init__(self) -> None: + self.last_usage_tokens = 0 + + def embed_documents(self, texts): + self.last_usage_tokens = sum(max(1, len(str(t)) // 4) for t in texts) + return [ + [1.0, 0.0] if "profile toolbar" in str(t).casefold() else [0.0, 1.0] + for t in texts + ] + + def embed_query(self, text): # noqa: ARG002 + self.last_usage_tokens = 1 + return [1.0, 0.0] + + +def _identity() -> CorpusIdentity: + return CorpusIdentity( + dataset_name="example/composition", + dataset_revision="rev-composition", + harness_commit="harness-composition-1", + tier="small", + domain="web", + ingest_config_digest="composition-test-v1", + ) + + +def _source(asset_root, *, trajectory_id, ordinal, goal, texts) -> TrajectorySource: + states = [] + for index, text in enumerate(texts): + screenshot = asset_root / f"{trajectory_id}-{index}.png" + screenshot.write_bytes(b"png" + hashlib.sha256(text.encode()).digest()) + states.append(TrajectoryState( + state_index=index, + step=index, + url=f"https://example.test/{trajectory_id}/{index}", + incoming_action=None if index == 0 else f"advance {index}", + thoughts=f"inspect state {index}", + text=text, + screenshot_path=screenshot, + )) + return TrajectorySource( + trajectory_id=trajectory_id, + ordinal=ordinal, + goal=goal, + start_url=f"https://example.test/{trajectory_id}", + outcome="completed", + states=tuple(states), + source_payload={"id": trajectory_id, "goal": goal}, + ) + + +_QUERY = "widget configuration export" + + +def _build_magnet_store(tmp_path: Path): + """A pure-lexical winner ('target') displaced by 12 semantic-top distractors.""" + asset_root = tmp_path / "assets" + asset_root.mkdir() + store = TrajectoryStore( + tmp_path / "lcm.db", + _identity(), + asset_root=asset_root, + embedding_provider=MagnetProvider(), + semantic_top_trajectories=12, + ) + # Strongest BM25 hit for the query, but NOT semantic-top (no 'profile toolbar'). + store.insert(_source( + asset_root, + trajectory_id="target", + ordinal=0, + goal="Configure the widget", + texts=("widget configuration export panel with all three terms",), + )) + order = ["target"] + # 12 distractors: semantic-top ('profile toolbar') + a weak lexical hit ('export'). + for index in range(12): + trajectory_id = f"distractor-{index:02d}" + store.insert(_source( + asset_root, + trajectory_id=trajectory_id, + ordinal=index + 1, + goal="Inspect profile toolbar", + texts=(f"profile toolbar export note {index}",), + )) + order.append(trajectory_id) + store.finalize(order) + store.build_semantic_index() + return store + + +def _delivered_trajectories(hits): + return [hit.trajectory_id for hit in hits] + + +def test_default_reproduces_the_magnet_displacement(tmp_path: Path): + # Baseline (byte-compat): the fused nucleus is monopolised by the semantic-top + # distractors and the pure-lexical winner is displaced out of delivery. + store = _build_magnet_store(tmp_path) + default_hits = store.query(_QUERY, limit=16, image_limit=0) + default_refs = [hit.exact_ref for hit in default_hits] + + assert "target" not in _delivered_trajectories(default_hits) + # The default kwargs are byte-identical to the explicit no-op knob values. + assert [h.exact_ref for h in store.query(_QUERY, limit=16, image_limit=0, lexical_floor=0)] == default_refs + assert [h.exact_ref for h in store.query(_QUERY, limit=16, image_limit=0, arm_quota=None)] == default_refs + + +def test_policy_a_lexical_floor_readmits_the_displaced_winner(tmp_path: Path): + store = _build_magnet_store(tmp_path) + assert "target" not in _delivered_trajectories(store.query(_QUERY, limit=16, image_limit=0)) + + floored = store.query(_QUERY, limit=16, image_limit=0, lexical_floor=1) + assert "target" in _delivered_trajectories(floored) + assert len({hit.exact_ref for hit in floored}) == len(floored) # no duplicates + + +def test_policy_d_arm_quota_readmits_the_displaced_winner(tmp_path: Path): + store = _build_magnet_store(tmp_path) + assert "target" not in _delivered_trajectories(store.query(_QUERY, limit=16, image_limit=0)) + + union = store.query(_QUERY, limit=16, image_limit=0, arm_quota=(6, 5)) + trajectories = _delivered_trajectories(union) + assert "target" in trajectories + # The semantic arm is still represented (quota union, not lexical-only). + assert any(t.startswith("distractor-") for t in trajectories) + assert len({hit.exact_ref for hit in union}) == len(union) + + +def test_merge_arms_dedups_and_backfills_without_wasting_quota(): + # Deterministic unit check of the arm merge: overlapping arms must not waste a + # quota slot on a duplicate, and a short arm is backfilled by the other. + def _rows(ids): + return [{"state_id": i, "trajectory_id": f"t{i}"} for i in ids] + + arm_lex = _rows([1, 2, 3, 4]) + arm_sem = _rows([2, 3, 5, 6]) # 2,3 overlap with lex + merged = TrajectoryStore._merge_arms(arm_lex, arm_sem, limit=6, q_lex=2, q_sem=2) + ids = [row["state_id"] for row in merged] + # round 1: lex[1,2] then sem[3,5] (2 is a deduped skip, not a wasted slot); + # round 2: lex[4] (3 deduped) then sem[6]. + assert ids == [1, 2, 3, 5, 4, 6] + assert len(ids) == len(set(ids)) + + +def test_policies_are_noops_without_semantic_ranks(tmp_path: Path): + # No embedding provider -> no semantic boost -> the fused order IS the lexical + # order, so both policies degenerate to the historical selection. + asset_root = tmp_path / "assets" + asset_root.mkdir() + store = TrajectoryStore( + tmp_path / "lcm.db", + _identity(), + asset_root=asset_root, + embedding_provider=None, + ) + store.insert(_source( + asset_root, + trajectory_id="target", + ordinal=0, + goal="Configure the widget", + texts=("widget configuration export panel",), + )) + store.finalize(["target"]) + baseline = [h.exact_ref for h in store.query(_QUERY, limit=8, image_limit=0)] + assert baseline + assert [h.exact_ref for h in store.query(_QUERY, limit=8, image_limit=0, lexical_floor=3)] == baseline + assert [h.exact_ref for h in store.query(_QUERY, limit=8, image_limit=0, arm_quota=(6, 5))] == baseline From 55351c1077a9344d22f95b344edd01ca7acad3ec Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 01:58:39 +0700 Subject: [PATCH 11/54] =?UTF-8?q?feat(composition):=20A+D=20hybrid=20(lexi?= =?UTF-8?q?cal=5Ffloor+arm=5Fquota)=20=E2=80=94=20QUARANTINED=20from=20sco?= =?UTF-8?q?red=20path=20per=20#127=20(2nd=20powered=20fail=20+7=20vs=20?= =?UTF-8?q?=E2=89=A5+8);=20preserved=20default-off;=20golden=20451/451=20b?= =?UTF-8?q?yte-identical?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- tests/test_trajectory_composition_policies.py | 44 +++++++++++++++++++ trajectory_store.py | 19 +++++++- 2 files changed, 62 insertions(+), 1 deletion(-) diff --git a/tests/test_trajectory_composition_policies.py b/tests/test_trajectory_composition_policies.py index 3eff31f88..62c741f2a 100644 --- a/tests/test_trajectory_composition_policies.py +++ b/tests/test_trajectory_composition_policies.py @@ -173,6 +173,50 @@ def _rows(ids): assert len(ids) == len(set(ids)) +def test_merge_arms_hybrid_floor_reserves_lexical_incumbents_first(): + # A+D hybrid (issue #127): floor_k reserves the top pure-lexical rows a slot + # BEFORE the quota round-robin, then the round-robin fills the rest. + # floor_k=0 (the default) is byte-identical to the pure Policy D merge. + def _rows(ids): + return [{"state_id": i, "trajectory_id": f"t{i}"} for i in ids] + + arm_lex = _rows([1, 2, 3, 4]) + arm_sem = _rows([2, 3, 5, 6]) # 2,3 overlap with lex + hybrid = TrajectoryStore._merge_arms( + arm_lex, arm_sem, limit=6, q_lex=2, q_sem=2, floor_k=1 + ) + ids = [row["state_id"] for row in hybrid] + assert ids[0] == 1 # top pure-lexical incumbent reserved by the floor first + assert ids == [1, 2, 3, 5, 6, 4] + assert len(ids) == len(set(ids)) + # floor_k=0 (and the omitted default) reproduce the pure Policy D bytes. + pure_d = [1, 2, 3, 5, 4, 6] + assert [r["state_id"] for r in TrajectoryStore._merge_arms( + arm_lex, arm_sem, limit=6, q_lex=2, q_sem=2)] == pure_d + assert [r["state_id"] for r in TrajectoryStore._merge_arms( + arm_lex, arm_sem, limit=6, q_lex=2, q_sem=2, floor_k=0)] == pure_d + + +def test_hybrid_floor_plus_quota_readmits_winner_and_composes(tmp_path: Path): + # The A+D hybrid composes: the lexical floor protects the displaced winner + # while the arm quota keeps the semantic arm represented; arm_quota with + # lexical_floor=0 is byte-identical to the pure Policy D delivery. + store = _build_magnet_store(tmp_path) + assert "target" not in _delivered_trajectories(store.query(_QUERY, limit=16, image_limit=0)) + + hybrid = store.query(_QUERY, limit=16, image_limit=0, lexical_floor=1, arm_quota=(6, 5)) + trajectories = _delivered_trajectories(hybrid) + assert "target" in trajectories # lexical incumbent protected by the floor + assert any(t.startswith("distractor-") for t in trajectories) # semantic arm kept + assert len({hit.exact_ref for hit in hybrid}) == len(hybrid) + + pure_d_refs = [h.exact_ref for h in store.query(_QUERY, limit=16, image_limit=0, arm_quota=(6, 5))] + assert [ + h.exact_ref + for h in store.query(_QUERY, limit=16, image_limit=0, lexical_floor=0, arm_quota=(6, 5)) + ] == pure_d_refs + + def test_policies_are_noops_without_semantic_ranks(tmp_path: Path): # No embedding provider -> no semantic boost -> the fused order IS the lexical # order, so both policies degenerate to the historical selection. diff --git a/trajectory_store.py b/trajectory_store.py index cb6462fe0..c3f446c64 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -1626,6 +1626,7 @@ def _merge_arms( limit: int, q_lex: int, q_sem: int, + floor_k: int = 0, ) -> list[sqlite3.Row]: """Policy D -- round-robin a pure-lexical arm and the semantic/fused arm into the nucleus by a ``q_lex:q_sem`` quota. @@ -1635,6 +1636,14 @@ def _merge_arms( quota (preserving the semantic gains). Deduped by ``state_id``; a short arm is backfilled by the other (a skipped duplicate does not consume a quota slot). Arm order is the deterministic tie-break. + + ``floor_k`` composes Policy A on top (the A+D hybrid, issue #127): the + top ``floor_k`` pure-lexical (BM25) states are reserved a nucleus slot + FIRST -- protecting the strongest lexical incumbents as Policy A does -- + and the quota round-robin then fills the remaining slots. ``floor_k == 0`` + (default) is byte-identical to the pure Policy D round-robin. The floor + is drawn from the head of ``arm_lex`` (already 5-per-trajectory + diversity-capped), matching ``_select_with_floor``'s guaranteed slots. """ selected: list[sqlite3.Row] = [] seen: set[int] = set() @@ -1654,6 +1663,9 @@ def _pull(arm: Sequence[sqlite3.Row], start: int, quota: int) -> int: return index lex_i = sem_i = 0 + floor_k = min(max(0, int(floor_k)), limit) + if floor_k: + lex_i = _pull(arm_lex, lex_i, floor_k) while len(selected) < limit and (lex_i < len(arm_lex) or sem_i < len(arm_sem)): next_lex = _pull(arm_lex, lex_i, q_lex) next_sem = _pull(arm_sem, sem_i, q_sem) @@ -1820,12 +1832,17 @@ def query( # pure-lexical arm and the semantic/fused arm into the nucleus by the # requested quota. Superset of Policy A; ``arm_quota is None`` # (default) is byte-identical to the historical selection below. + # When ``lexical_floor > 0`` is ALSO supplied this becomes the A+D + # hybrid: the top ``lexical_floor`` pure-BM25 incumbents are reserved + # a slot first, then the quota round-robin fills the rest + # (``lexical_floor == 0`` reproduces the pure Policy D bytes). q_lex = max(0, int(arm_quota[0])) q_sem = max(0, int(arm_quota[1])) arm_lex = self._select_diverse(global_rows, nucleus_limit) arm_sem = self._select_diverse(rows, nucleus_limit) selected = self._merge_arms( - arm_lex, arm_sem, nucleus_limit, q_lex, q_sem + arm_lex, arm_sem, nucleus_limit, q_lex, q_sem, + floor_k=lexical_floor, ) elif lexical_floor > 0: # Policy A (candidate-composition repair, issue #127): guarantee the From 54379ecd183941f902c82aea4a11f4a7fbd24217 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 02:58:55 +0700 Subject: [PATCH 12/54] trajectory: H5(b) lexical-seed adjacency pool-expansion (#135) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Promote adjacency from a delivery-only ±1 backfill to a POOL-stage expansion: for every lexical seed hit in the candidate pool, the states at sequence_ordinal ±1..adjacency_radius within the same source are admitted as a QUOTA-CAPPED ADDITIVE arm through the existing _merge_arms machinery (q_lex=len(rows), q_sem=adjacency_quota), giving a non-lexical recall path for states that carry no query term of their own. Controls (anti-magnet, per H5-design-options option (b) + SPEC-H5b): - strictly additive tail: expanded neighbors earn NO semantic boost and no BM25 rank; nucleus selection -- and therefore delivery -- changes only when the ranked pool cannot fill the nucleus (gate (ii) additive proof); - deterministic distance-major / seed-pool-rank / ordinal arm order; - 5-per-trajectory diversity cap at selection untouched; - adjacency_radius=0 / adjacency_quota=0 defaults skip the path entirely (current bytes; telemetry key present only when active); - batched row-value IN lookups on the (source_id, sequence_ordinal) index. tests: synthetic invisible-neighbor corpus -- default byte-identity incl. half-open knobs, pool entry, radius bound + arm order, quota cap, pool dedup, delivery-unchanged on a full (magnet) pool, selection cap, tail-only placement. Full suite: failure set identical to base 65f679b under the same invocation (pre-existing env-dependent failures only). --- tests/test_trajectory_adjacency_expansion.py | 345 +++++++++++++++++++ trajectory_store.py | 134 +++++++ 2 files changed, 479 insertions(+) create mode 100644 tests/test_trajectory_adjacency_expansion.py diff --git a/tests/test_trajectory_adjacency_expansion.py b/tests/test_trajectory_adjacency_expansion.py new file mode 100644 index 000000000..1194b1d7e --- /dev/null +++ b/tests/test_trajectory_adjacency_expansion.py @@ -0,0 +1,345 @@ +"""H5(b) lexical-seed adjacency pool-expansion (issue #135). + +These exercise the candidate-RECALL pathology on a synthetic corpus: a target +state carries NO query term of its own (lexically invisible) while a sibling +state of the same trajectory seeds, so the target is unreachable at the pool +stage under current bytes. The ``adjacency_radius``/``adjacency_quota`` knob +must pull the invisible neighbor INTO the state pool as a quota-capped +additive arm, while the defaults must reproduce the pre-expansion pool and +delivery byte-for-byte, delivery must stay unchanged when the ranked pool +already fills the nucleus (additive-only proof), and the 5-per-trajectory +diversity cap at selection must be preserved. +""" + +from __future__ import annotations + +import hashlib +from pathlib import Path + +from hermes_lcm.trajectory_store import ( + CorpusIdentity, + TrajectorySource, + TrajectoryState, + TrajectoryStore, +) + + +class MagnetProvider: + """Ranks any trajectory whose text mentions 'profile toolbar' as semantically + top, regardless of the query -- the coarse whole-trajectory 'magnet'.""" + + provider_id = "fake" + model_id = "fake-trajectory-v1" + dim = 2 + + def __init__(self) -> None: + self.last_usage_tokens = 0 + + def embed_documents(self, texts): + self.last_usage_tokens = sum(max(1, len(str(t)) // 4) for t in texts) + return [ + [1.0, 0.0] if "profile toolbar" in str(t).casefold() else [0.0, 1.0] + for t in texts + ] + + def embed_query(self, text): # noqa: ARG002 + self.last_usage_tokens = 1 + return [1.0, 0.0] + + +def _identity() -> CorpusIdentity: + return CorpusIdentity( + dataset_name="example/adjacency", + dataset_revision="rev-adjacency", + harness_commit="harness-adjacency-1", + tier="small", + domain="web", + ingest_config_digest="adjacency-test-v1", + ) + + +def _source(asset_root, *, trajectory_id, ordinal, goal, texts) -> TrajectorySource: + states = [] + for index, text in enumerate(texts): + screenshot = asset_root / f"{trajectory_id}-{index}.png" + screenshot.write_bytes(b"png" + hashlib.sha256(text.encode()).digest()) + states.append(TrajectoryState( + state_index=index, + step=index, + url=f"https://example.test/{trajectory_id}/{index}", + incoming_action=None if index == 0 else f"advance {index}", + thoughts=f"inspect state {index}", + text=text, + screenshot_path=screenshot, + )) + return TrajectorySource( + trajectory_id=trajectory_id, + ordinal=ordinal, + goal=goal, + start_url=f"https://example.test/{trajectory_id}", + outcome="completed", + states=tuple(states), + source_payload={"id": trajectory_id, "goal": goal}, + ) + + +_QUERY = "widget configuration export" + + +def _state_id(store, trajectory_id: str, state_index: int) -> int: + row = store._conn.execute( + """ + SELECT s.state_id FROM lcm_trajectory_states s + JOIN lcm_trajectory_sources src ON src.source_id = s.source_id + WHERE src.trajectory_id = ? AND s.state_index = ? + """, + (trajectory_id, state_index), + ).fetchone() + assert row is not None, f"unknown state {trajectory_id}/{state_index}" + return int(row[0]) + + +def _pool_state_ids(store) -> set[int]: + telemetry = store.last_query_telemetry() + return {int(item["state_id"]) for item in telemetry["state_candidate_pool"]} + + +def _admitted(store) -> list[dict[str, int]]: + telemetry = store.last_query_telemetry() + expansion = telemetry.get("adjacency_expansion") + return list(expansion["admitted"]) if expansion else [] + + +def _build_invisible_neighbor_store(tmp_path: Path): + """One trajectory whose answer state is lexically INVISIBLE (no query + term) while its predecessor seeds; plus a second lexical trajectory.""" + asset_root = tmp_path / "assets" + asset_root.mkdir() + store = TrajectoryStore(tmp_path / "lcm.db", _identity(), asset_root=asset_root) + store.insert(_source( + asset_root, + trajectory_id="answerpath", + ordinal=0, + goal="Update the account settings", + texts=( + "navigate homepage dashboard overview", # 0: invisible, -1 of seed + "widget configuration export panel form", # 1: the lexical seed + "save success banner shows view details link", # 2: invisible ANSWER state + "logout footer copyright notice", # 3: invisible, +2 of seed + ), + )) + store.insert(_source( + asset_root, + trajectory_id="othertask", + ordinal=1, + goal="Export the report", + texts=("report export toolbar button",), + )) + store.finalize(["answerpath", "othertask"]) + return store + + +# --- default-off byte-identity ---------------------------------------------- + +def test_defaults_reproduce_current_bytes(tmp_path): + store = _build_invisible_neighbor_store(tmp_path) + baseline = store.query(_QUERY, image_limit=0) + baseline_telemetry = store.last_query_telemetry() + explicit_off = store.query( + _QUERY, image_limit=0, adjacency_radius=0, adjacency_quota=0, + ) + off_telemetry = store.last_query_telemetry() + assert [h.exact_ref for h in explicit_off] == [h.exact_ref for h in baseline] + assert off_telemetry == baseline_telemetry + # No telemetry key leaks into the default payload (frozen-run byte parity). + assert "adjacency_expansion" not in baseline_telemetry + # Half-open knobs (radius without quota and vice versa) are also OFF. + for kwargs in ({"adjacency_radius": 2}, {"adjacency_quota": 8}): + hits = store.query(_QUERY, image_limit=0, **kwargs) + assert [h.exact_ref for h in hits] == [h.exact_ref for h in baseline] + assert "adjacency_expansion" not in store.last_query_telemetry() + + +# --- the core recall mechanism ---------------------------------------------- + +def test_expansion_pulls_lexically_invisible_neighbor_into_pool(tmp_path): + store = _build_invisible_neighbor_store(tmp_path) + answer = _state_id(store, "answerpath", 2) + store.query(_QUERY, image_limit=0) + assert answer not in _pool_state_ids(store), "answer state must start invisible" + store.query(_QUERY, image_limit=0, adjacency_radius=1, adjacency_quota=8) + assert answer in _pool_state_ids(store) + seed = _state_id(store, "answerpath", 1) + by_state = {entry["state_id"]: entry for entry in _admitted(store)} + assert by_state[answer] == {"state_id": answer, "seed_state_id": seed, "distance": 1} + + +def test_radius_bounds_the_reach_and_orders_distance_major(tmp_path): + store = _build_invisible_neighbor_store(tmp_path) + far = _state_id(store, "answerpath", 3) # +2 from the seed + store.query(_QUERY, image_limit=0, adjacency_radius=1, adjacency_quota=8) + assert far not in _pool_state_ids(store) + store.query(_QUERY, image_limit=0, adjacency_radius=2, adjacency_quota=8) + admitted = _admitted(store) + assert far in {entry["state_id"] for entry in admitted} + distances = [entry["distance"] for entry in admitted] + assert distances == sorted(distances), "arm order must be distance-major" + + +def test_quota_caps_admissions(tmp_path): + store = _build_invisible_neighbor_store(tmp_path) + store.query(_QUERY, image_limit=0, adjacency_radius=2, adjacency_quota=1) + admitted = _admitted(store) + assert len(admitted) == 1 + # Distance-major: the single slot goes to a distance-1 neighbor. + assert admitted[0]["distance"] == 1 + + +def test_pool_incumbents_are_not_readmitted(tmp_path): + """A neighbor that already entered the pool lexically is never duplicated.""" + asset_root = tmp_path / "assets" + asset_root.mkdir() + store = TrajectoryStore(tmp_path / "lcm.db", _identity(), asset_root=asset_root) + store.insert(_source( + asset_root, + trajectory_id="allmatch", + ordinal=0, + goal="Finish the setup flow", + texts=( + "widget configuration export intro", + "widget configuration export detail", + "plain closing remark", # only non-matching state + ), + )) + store.finalize(["allmatch"]) + store.query(_QUERY, image_limit=0, adjacency_radius=2, adjacency_quota=8) + admitted = _admitted(store) + only = _state_id(store, "allmatch", 2) + assert [entry["state_id"] for entry in admitted] == [only] + pool = [ + int(item["state_id"]) + for item in store.last_query_telemetry()["state_candidate_pool"] + ] + assert len(pool) == len(set(pool)), "pool must stay duplicate-free" + + +# --- anti-filler / additive-only controls ------------------------------------- + +def test_delivery_unchanged_when_ranked_pool_fills_nucleus(tmp_path): + """Additive-only proof on a full pool: the expanded states may enter the + POOL but must not displace any delivered nucleus/backfill incumbent.""" + asset_root = tmp_path / "assets" + asset_root.mkdir() + store = TrajectoryStore( + tmp_path / "lcm.db", + _identity(), + asset_root=asset_root, + embedding_provider=MagnetProvider(), + semantic_top_trajectories=12, + ) + store.insert(_source( + asset_root, + trajectory_id="target", + ordinal=0, + goal="Set up the dashboard gadget", + texts=( + "widget configuration export panel with all three terms", + "invisible aftermath confirmation banner", + ), + )) + order = ["target"] + for index in range(12): + trajectory_id = f"distractor-{index:02d}" + store.insert(_source( + asset_root, + trajectory_id=trajectory_id, + ordinal=index + 1, + goal="Inspect profile toolbar", + texts=( + f"profile toolbar export note {index}", + f"quiet interstitial screen {index}", + ), + )) + order.append(trajectory_id) + store.finalize(order) + store.build_semantic_index() + + baseline = [h.exact_ref for h in store.query(_QUERY, image_limit=0)] + expanded = [ + h.exact_ref + for h in store.query( + _QUERY, image_limit=0, adjacency_radius=1, adjacency_quota=16, + ) + ] + assert expanded == baseline + admitted = _admitted(store) + assert admitted, "the pool itself must still gain adjacency entries" + # The magnet's neighbors outrank the target's in the (seed-strength + # ordered) arm, but a quota that clears the 12 distractors still pulls + # the lexically invisible target state into the pool. + invisible = _state_id(store, "target", 1) + assert invisible in {entry["state_id"] for entry in admitted} + + +def test_five_per_trajectory_cap_preserved_at_selection(tmp_path): + asset_root = tmp_path / "assets" + asset_root.mkdir() + store = TrajectoryStore(tmp_path / "lcm.db", _identity(), asset_root=asset_root) + store.insert(_source( + asset_root, + trajectory_id="longtask", + ordinal=0, + goal="Complete the long procedure", + texts=tuple( + "widget configuration export summary step" + if index == 3 + else f"quiet screen number {index}" + for index in range(8) + ), + )) + store.insert(_source( + asset_root, + trajectory_id="filler", + ordinal=1, + goal="Export widgets", + texts=tuple(f"widget export list page {index}" for index in range(6)), + )) + store.finalize(["longtask", "filler"]) + hits = store.query( + _QUERY, + image_limit=0, + include_adjacent=False, # nucleus only: isolate the selection cap + adjacency_radius=8, + adjacency_quota=32, + ) + per_trajectory: dict[str, int] = {} + for hit in hits: + per_trajectory[hit.trajectory_id] = per_trajectory.get(hit.trajectory_id, 0) + 1 + assert per_trajectory.get("longtask", 0) <= 5 + # ...even though MORE than 5 longtask states were admitted to the pool. + longtask_admitted = [ + entry + for entry in _admitted(store) + if entry["state_id"] in { + _state_id(store, "longtask", index) for index in range(8) + } + ] + assert len(longtask_admitted) == 7 + + +def test_expanded_states_selected_only_from_the_tail(tmp_path): + """Expanded neighbors earn no semantic/BM25 rank: every ranked pool row + still precedes every admitted adjacency row in the candidate pool.""" + store = _build_invisible_neighbor_store(tmp_path) + store.query(_QUERY, image_limit=0, adjacency_radius=2, adjacency_quota=8) + telemetry = store.last_query_telemetry() + pool = [int(item["state_id"]) for item in telemetry["state_candidate_pool"]] + admitted_ids = {entry["state_id"] for entry in _admitted(store)} + ranked_positions = [ + index for index, state_id in enumerate(pool) if state_id not in admitted_ids + ] + admitted_positions = [ + index for index, state_id in enumerate(pool) if state_id in admitted_ids + ] + assert admitted_positions, "expanded states must be visible in the pool" + assert max(ranked_positions) < min(admitted_positions) diff --git a/trajectory_store.py b/trajectory_store.py index c3f446c64..1491e1d73 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -39,6 +39,7 @@ _MAX_RESULTS = 24 _MAX_IMAGES = 8 _MAX_QUERY_TEXT_CHARS = 8_000 +_MAX_ADJACENCY_RADIUS = 8 _MAX_SOURCE_JSON_CHARS = 16_000_000 _MAX_TEXT_CHARS = 2_000_000 _MAX_SEMANTIC_DOCUMENT_CHARS = 48_000 @@ -1705,6 +1706,87 @@ def _fts_rows( tuple(params), ).fetchall() + def _adjacency_expansion_arm( + self, + seed_rows: Sequence[sqlite3.Row], + radius: int, + ) -> list[tuple[sqlite3.Row, int, int]]: + """H5(b) pool-expansion arm (issue #135): sequence neighbors of the + lexical seed hits, as ``(row, seed_state_id, distance)`` triples. + + Every pool row IS a lexical seed (it entered via ``global_rows`` / + ``scoped_rows`` FTS), so the seeds are exactly ``seed_rows`` in pool + (fused-rank) order. For each seed, the states at ``sequence_ordinal + +/- 1..radius`` WITHIN THE SAME SOURCE are candidates, giving a + non-lexical recall path: a target state with no query-term match of + its own is reachable when any state of its trajectory seeds. + + Deterministic arm order is DISTANCE-major, then seed pool rank, then + ordinal ascending (``-d`` before ``+d``): a +/-1 neighbor of any seed + outranks a +/-2 neighbor of a stronger seed, mirroring the + ``ORDER BY ABS(...)`` discipline of the delivery-stage adjacency + backfill. States already in the pool are excluded; expanded neighbors + earn NO semantic boost and no BM25 rank of their own (anti-magnet + control -- they are admitted by the caller only through the + quota-capped ``_merge_arms`` tail). + """ + pool_ids = {int(row["state_id"]) for row in seed_rows} + positions: list[tuple[int, int]] = [] + wanted: set[tuple[int, int]] = set() + occupied = { + (int(row["source_id"]), int(row["sequence_ordinal"])) + for row in seed_rows + } + for row in seed_rows: + source_id = int(row["source_id"]) + ordinal = int(row["sequence_ordinal"]) + for distance in range(1, radius + 1): + for neighbor in (ordinal - distance, ordinal + distance): + key = (source_id, neighbor) + if neighbor < 0 or key in occupied or key in wanted: + continue + wanted.add(key) + positions.append(key) + row_by_position: dict[tuple[int, int], sqlite3.Row] = {} + chunk = 400 # 2 bound params per pair; stay far below SQLite limits + for start in range(0, len(positions), chunk): + batch = positions[start : start + chunk] + values = ",".join("(?,?)" for _ in batch) + params: list[int] = [] + for source_id, neighbor in batch: + params.extend((source_id, neighbor)) + fetched = self._conn.execute( + f""" + SELECT s.*, src.trajectory_id, src.goal, src.outcome, src.ordinal, + a.relative_path, a.sha256 AS asset_sha256, 0.0 AS rank + FROM lcm_trajectory_states s + JOIN lcm_trajectory_sources src ON src.source_id = s.source_id + LEFT JOIN lcm_trajectory_assets a ON a.state_id = s.state_id + WHERE (s.source_id, s.sequence_ordinal) IN (VALUES {values}) + """, + params, + ).fetchall() + for fetched_row in fetched: + row_by_position[ + (int(fetched_row["source_id"]), int(fetched_row["sequence_ordinal"])) + ] = fetched_row + arm: list[tuple[sqlite3.Row, int, int]] = [] + emitted: set[int] = set(pool_ids) + for distance in range(1, radius + 1): + for row in seed_rows: + source_id = int(row["source_id"]) + ordinal = int(row["sequence_ordinal"]) + for neighbor in (ordinal - distance, ordinal + distance): + neighbor_row = row_by_position.get((source_id, neighbor)) + if neighbor_row is None: + continue + state_id = int(neighbor_row["state_id"]) + if state_id in emitted: + continue + emitted.add(state_id) + arm.append((neighbor_row, int(row["state_id"]), distance)) + return arm + def query( self, query: str, @@ -1716,6 +1798,8 @@ def query( text_char_limit: int = 2_000, lexical_floor: int = 0, arm_quota: tuple[int, int] | None = None, + adjacency_radius: int = 0, + adjacency_quota: int = 0, ) -> tuple[TrajectoryHit, ...]: if self.status != "complete": raise CorpusIdentityError("trajectory corpus must be finalized before query") @@ -1723,6 +1807,8 @@ def query( limit = min(max(1, int(limit)), _MAX_RESULTS) image_limit = min(max(0, int(image_limit)), _MAX_IMAGES) lexical_floor = min(max(0, int(lexical_floor)), _MAX_RESULTS) + adjacency_radius = min(max(0, int(adjacency_radius)), _MAX_ADJACENCY_RADIUS) + adjacency_quota = min(max(0, int(adjacency_quota)), _MAX_CANDIDATES) text_char_limit = min( max(256, int(text_char_limit)), _MAX_QUERY_TEXT_CHARS, @@ -1825,6 +1911,46 @@ def query( for row in rows } + # H5(b) lexical-seed adjacency pool-expansion (issue #135): pull the + # sequence neighbors of the lexical seed hits INTO the state pool, + # pre-selection, as a QUOTA-CAPPED ADDITIVE arm through the existing + # ``_merge_arms`` machinery. The arm is appended strictly AFTER the + # ranked pool (no semantic boost, no BM25 rank of its own), so the + # nucleus selection -- and therefore delivery -- only changes when the + # ranked pool alone cannot fill the nucleus; the 5-per-trajectory + # diversity cap at selection is untouched. ``adjacency_radius == 0`` + # or ``adjacency_quota == 0`` (defaults) skip this path entirely and + # reproduce current bytes. + adjacency_admitted: list[dict[str, int]] = [] + if adjacency_radius > 0 and adjacency_quota > 0 and rows: + arm_triples = self._adjacency_expansion_arm(rows, adjacency_radius) + if arm_triples: + arm_adjacent = [triple[0] for triple in arm_triples] + seed_by_state = { + int(triple[0]["state_id"]): (triple[1], triple[2]) + for triple in arm_triples + } + expanded = self._merge_arms( + rows, + arm_adjacent, + len(rows) + adjacency_quota, + len(rows), + adjacency_quota, + ) + for row in expanded[len(rows):]: + state_id = int(row["state_id"]) + seed_state_id, distance = seed_by_state[state_id] + candidate_kind[state_id] = "adjacent" + candidate_score[state_id] = ( + candidate_score[seed_state_id] + 0.000001 * distance + ) + adjacency_admitted.append({ + "state_id": state_id, + "seed_state_id": seed_state_id, + "distance": distance, + }) + rows = expanded + adjacent_reserve = min(6, limit // 3) if include_adjacent else 0 nucleus_limit = max(1, limit - adjacent_reserve) if arm_quota is not None: @@ -1938,6 +2064,14 @@ def query( ], "delivered_evidence_refs": [hit.exact_ref for hit in hits], } + if adjacency_radius > 0 and adjacency_quota > 0: + # Present only when the H5(b) knob is active so the default + # telemetry payload stays byte-identical (golden 451/451). + self._last_query_telemetry["adjacency_expansion"] = { + "radius": adjacency_radius, + "quota": adjacency_quota, + "admitted": adjacency_admitted, + } return tuple(hits) def resolve_exact_ref(self, exact_ref: str) -> TrajectoryHit: From d3517d53927447a200692f93c4edf197336ac847 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 03:30:43 +0700 Subject: [PATCH 13/54] trajectory: fetch full rows only for quota-admitted expansion states (#135) The arm probe is now index-only (state_id/source_id/sequence_ordinal); the candidate-shaped full rows (large state text) are fetched solely for the <=quota admitted states. Measured on the frozen enterprise corpus: ~400-pair full-row probe cost ~63ms/chunk vs 0.07ms light probe + ~5ms admitted fetch -- keeps the knob inside the p95 +10% latency gate (iv). Behavior unchanged (41 trajectory-suite tests green; golden 451/451 re-verified). --- trajectory_store.py | 71 ++++++++++++++++++++++++++++++--------------- 1 file changed, 48 insertions(+), 23 deletions(-) diff --git a/trajectory_store.py b/trajectory_store.py index 1491e1d73..56ba517a8 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -1710,9 +1710,9 @@ def _adjacency_expansion_arm( self, seed_rows: Sequence[sqlite3.Row], radius: int, - ) -> list[tuple[sqlite3.Row, int, int]]: + ) -> list[tuple[int, int, int]]: """H5(b) pool-expansion arm (issue #135): sequence neighbors of the - lexical seed hits, as ``(row, seed_state_id, distance)`` triples. + lexical seed hits, as ``(state_id, seed_state_id, distance)`` triples. Every pool row IS a lexical seed (it entered via ``global_rows`` / ``scoped_rows`` FTS), so the seeds are exactly ``seed_rows`` in pool @@ -1729,6 +1729,11 @@ def _adjacency_expansion_arm( earn NO semantic boost and no BM25 rank of their own (anti-magnet control -- they are admitted by the caller only through the quota-capped ``_merge_arms`` tail). + + Returns LIGHTWEIGHT id triples (a batched index-only probe): the full + rows -- state text is large -- are fetched by the caller for the + ADMITTED quota subset only, keeping the expansion inside the latency + budget (gate iv). """ pool_ids = {int(row["state_id"]) for row in seed_rows} positions: list[tuple[int, int]] = [] @@ -1747,7 +1752,7 @@ def _adjacency_expansion_arm( continue wanted.add(key) positions.append(key) - row_by_position: dict[tuple[int, int], sqlite3.Row] = {} + id_by_position: dict[tuple[int, int], int] = {} chunk = 400 # 2 bound params per pair; stay far below SQLite limits for start in range(0, len(positions), chunk): batch = positions[start : start + chunk] @@ -1757,36 +1762,51 @@ def _adjacency_expansion_arm( params.extend((source_id, neighbor)) fetched = self._conn.execute( f""" - SELECT s.*, src.trajectory_id, src.goal, src.outcome, src.ordinal, - a.relative_path, a.sha256 AS asset_sha256, 0.0 AS rank + SELECT s.state_id, s.source_id, s.sequence_ordinal FROM lcm_trajectory_states s - JOIN lcm_trajectory_sources src ON src.source_id = s.source_id - LEFT JOIN lcm_trajectory_assets a ON a.state_id = s.state_id WHERE (s.source_id, s.sequence_ordinal) IN (VALUES {values}) """, params, ).fetchall() for fetched_row in fetched: - row_by_position[ + id_by_position[ (int(fetched_row["source_id"]), int(fetched_row["sequence_ordinal"])) - ] = fetched_row - arm: list[tuple[sqlite3.Row, int, int]] = [] + ] = int(fetched_row["state_id"]) + arm: list[tuple[int, int, int]] = [] emitted: set[int] = set(pool_ids) for distance in range(1, radius + 1): for row in seed_rows: source_id = int(row["source_id"]) ordinal = int(row["sequence_ordinal"]) for neighbor in (ordinal - distance, ordinal + distance): - neighbor_row = row_by_position.get((source_id, neighbor)) - if neighbor_row is None: - continue - state_id = int(neighbor_row["state_id"]) - if state_id in emitted: + state_id = id_by_position.get((source_id, neighbor)) + if state_id is None or state_id in emitted: continue emitted.add(state_id) - arm.append((neighbor_row, int(row["state_id"]), distance)) + arm.append((state_id, int(row["state_id"]), distance)) return arm + def _state_rows_by_ids(self, state_ids: Sequence[int]) -> dict[int, sqlite3.Row]: + """Full candidate-shaped rows (text, source join, asset join) for the + given state ids -- the same column shape as ``_fts_rows`` with a + placeholder rank, fetched only for the quota-admitted expansion + states.""" + if not state_ids: + return {} + placeholders = ",".join("?" for _ in state_ids) + rows = self._conn.execute( + f""" + SELECT s.*, src.trajectory_id, src.goal, src.outcome, src.ordinal, + a.relative_path, a.sha256 AS asset_sha256, 0.0 AS rank + FROM lcm_trajectory_states s + JOIN lcm_trajectory_sources src ON src.source_id = s.source_id + LEFT JOIN lcm_trajectory_assets a ON a.state_id = s.state_id + WHERE s.state_id IN ({placeholders}) + """, + [int(state_id) for state_id in state_ids], + ).fetchall() + return {int(row["state_id"]): row for row in rows} + def query( self, query: str, @@ -1925,21 +1945,26 @@ def query( if adjacency_radius > 0 and adjacency_quota > 0 and rows: arm_triples = self._adjacency_expansion_arm(rows, adjacency_radius) if arm_triples: - arm_adjacent = [triple[0] for triple in arm_triples] seed_by_state = { - int(triple[0]["state_id"]): (triple[1], triple[2]) - for triple in arm_triples + state_id: (seed_state_id, distance) + for state_id, seed_state_id, distance in arm_triples } - expanded = self._merge_arms( + arm_adjacent = [ + {"state_id": state_id} for state_id, _seed, _dist in arm_triples + ] + merged = self._merge_arms( rows, - arm_adjacent, + arm_adjacent, # type: ignore[arg-type] # only "state_id" is read len(rows) + adjacency_quota, len(rows), adjacency_quota, ) - for row in expanded[len(rows):]: - state_id = int(row["state_id"]) + admitted_ids = [int(row["state_id"]) for row in merged[len(rows):]] + full_by_id = self._state_rows_by_ids(admitted_ids) + expanded = list(rows) + for state_id in admitted_ids: seed_state_id, distance = seed_by_state[state_id] + expanded.append(full_by_id[state_id]) candidate_kind[state_id] = "adjacent" candidate_score[state_id] = ( candidate_score[seed_state_id] + 0.000001 * distance From 5f92ee474df91d042029c8ee63745c8b414abdef Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 03:30:43 +0700 Subject: [PATCH 14/54] bench: provider-free H5(b) recall replay harness (#135) Extends the H3.1 composition replay instrument: same frozen assets, same injected semantic ranks, same mandatory golden gate (451/451 before any sweep). Sweeps adjacency radius x quota under TWO compositions (base + quarantined hybrid lexical_floor=1/arm_quota=(7,4)) and reports per knob: pool-entry recovery over the 30 verified H32 targets (h5-targets.json; EXACT pool membership, not 64-truncated telemetry), Delivered-Recall@16 vs the same composition's no-adjacency delivery, the 8 loss-ids' delivered-set anti-filler check, preservation disturbance over the 154 stable-correct questions, and paired-pass p95 latency (external-drive I/O variance makes unpaired reads incomparable). Also emits the zero/weak-seed split of the 30. The component gate is FROZEN: this instrument does not pick the knob. --- benchmarking/h5_recall_replay.py | 423 +++++++++++++++++++++++++++++++ 1 file changed, 423 insertions(+) create mode 100644 benchmarking/h5_recall_replay.py diff --git a/benchmarking/h5_recall_replay.py b/benchmarking/h5_recall_replay.py new file mode 100644 index 000000000..7a366f358 --- /dev/null +++ b/benchmarking/h5_recall_replay.py @@ -0,0 +1,423 @@ +#!/usr/bin/env python3 +"""Provider-free replay harness for the H5(b) adjacency pool-expansion (#135). + +Extends the H3.1 composition replay instrument (``h3_composition_replay``): +same frozen assets, same injected semantic ranks, same golden gate. Sweeps the +``adjacency_radius x adjacency_quota`` knob grid under TWO arm compositions -- +``base`` (shipping defaults) and ``hybrid`` (the quarantined A+D composition +``lexical_floor=1, arm_quota=(7,4)``) -- and measures, per knob x composition: + + POOL-ENTRY RECOVERY -- of the 30 verified H32 NOT_READMITTED cases + (h5-targets.json), how many gain a NEW target state + in the candidate pool (pinned targets: a pinned + state newly present; trajectory targets: a + trajectory with ZERO default-pool states gains one). + Pool membership is computed EXACTLY (global + scoped + FTS union + admitted expansion states), not from the + 64-truncated telemetry. Knob-level (the pool is + composition-independent). + DELIVERED-RECALL@16 -- same rule against the delivered refs, vs the SAME + composition's no-adjacency delivery. + ANTI-FILLER (gate ii)-- the 8 H32 loss-ids' delivered sets must be UNCHANGED + vs the same composition's no-adjacency delivery. + PRESERVATION -- stable-correct questions whose same-composition + baseline delivered refs drop under the knob. + LATENCY (gate iv) -- p95 over a 50-query replay vs the same composition + without adjacency (frozen gate reads the base lane). + +Also emits the ZERO-SEED / WEAK-SEED split of the 30 (which target +trajectories have no lexical match at all vs a match outside the pool vs a +seed already in the pool -- the adjacency mechanism's reachability boundary). + +The component gate is FROZEN (SPEC-H5b): this instrument reports the full +honest table; it does NOT pick the shipping knob. +""" +from __future__ import annotations + +import argparse +import json +import sys +import time +from pathlib import Path +from typing import Any +from urllib.parse import unquote + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from benchmarking.h3_composition_replay import ( # noqa: E402 + _H1, + _H31, + _REF_RE, + _RUN_ROOT, + ReplayContext, + golden_gate, + measure_latency, +) + +_H5_TARGETS = _H1 / "h5-targets.json" +_LOSS_8 = [ + ("web", "41bb40df"), ("web", "0738c1ac"), ("enterprise", "ccf29914"), + ("enterprise", "2721ca7f"), ("web", "896017e0"), ("enterprise", "0b50ca0d"), + ("enterprise", "2e90de97"), ("enterprise", "58c34839"), +] +_COMPOSITIONS: dict[str, dict[str, Any]] = { + "base": {}, + "hybrid": {"lexical_floor": 1, "arm_quota": (7, 4)}, +} + + +def _ref_pairs(refs: list[str]) -> set[tuple[str, int]]: + """Delivered refs -> {(trajectory_id, state_index)}.""" + out: set[tuple[str, int]] = set() + for ref in refs: + match = _REF_RE.match(ref) + if match: + out.add((unquote(match.group("traj")), int(match.group("state")))) + return out + + +class H5Context: + """Target-aware wrapper over the H3.1 ReplayContext.""" + + def __init__(self, ctx: ReplayContext, targets_path: Path) -> None: + self.ctx = ctx + payload = json.loads(targets_path.read_text()) + self.cases: list[dict[str, Any]] = payload["cases"] + assert len(self.cases) == int(payload["denominator"]) == 30 + self._traj_states: dict[tuple[str, str], dict[int, int]] = {} + self._pool_cache: dict[str, set[int]] = {} + + # -- corpus lookups -------------------------------------------------- + def traj_states(self, domain: str, trajectory_id: str) -> dict[int, int]: + """{state_index: state_id} for one trajectory.""" + key = (domain, trajectory_id) + if key not in self._traj_states: + store = self.ctx.stores[domain] + rows = store._conn.execute( + """SELECT s.state_index, s.state_id + FROM lcm_trajectory_states s + JOIN lcm_trajectory_sources src ON src.source_id = s.source_id + WHERE src.trajectory_id = ?""", + (trajectory_id,), + ).fetchall() + assert rows, f"{domain}/{trajectory_id} missing from frozen DB" + self._traj_states[key] = {int(r[0]): int(r[1]) for r in rows} + return self._traj_states[key] + + def traj_source_id(self, domain: str, trajectory_id: str) -> int: + store = self.ctx.stores[domain] + row = store._conn.execute( + "SELECT source_id FROM lcm_trajectory_sources WHERE trajectory_id = ?", + (trajectory_id,), + ).fetchone() + return int(row[0]) + + # -- exact pool membership -------------------------------------------- + def default_pool_ids(self, qid: str) -> set[int]: + """The EXACT default candidate pool: global FTS top-128 union scoped + FTS top-128 within the injected semantic-top sources (mirrors + ``TrajectoryStore.query`` pool construction; membership only).""" + if qid in self._pool_cache: + return self._pool_cache[qid] + domain, text = self.ctx.questions[qid] + store = self.ctx.stores[domain] + expression = store._fts_expression(text) + ids: set[int] = set() + if expression: + ids = {int(r["state_id"]) for r in store._fts_rows(expression, 128)} + ranks = self.ctx._injected_ranks(qid) + if ranks: + scoped = store._fts_rows( + expression, 128, + source_ids=[source_id for source_id, _score in ranks], + ) + ids |= {int(r["state_id"]) for r in scoped} + self._pool_cache[qid] = ids + return ids + + def deliver_and_admitted( + self, qid: str, **kwargs: Any + ) -> tuple[list[str], set[int]]: + """Delivered refs + the adjacency-admitted state ids (telemetry).""" + domain, _text = self.ctx.questions[qid] + delivered = self.ctx.deliver(qid, **kwargs) + telemetry = self.ctx.stores[domain].last_query_telemetry() + expansion = telemetry.get("adjacency_expansion") or {} + admitted = {int(e["state_id"]) for e in expansion.get("admitted", [])} + return delivered, admitted + + # -- per-case classification ------------------------------------------- + def seed_split(self) -> dict[str, Any]: + """Zero-seed / weak-seed / seeded split of the 30 target cases.""" + per_case: dict[str, str] = {} + for case in self.cases: + if not case["targets"]: + per_case[case["qid"]] = "no_target" + continue + domain = case["domain"] + qid = case["qid"] + store = self.ctx.stores[domain] + _dom, text = self.ctx.questions[qid] + expression = store._fts_expression(text) + pool = self.default_pool_ids(qid) + any_match = False + any_pooled = False + for target in case["targets"]: + states = self.traj_states(domain, target["trajectory_id"]) + if set(states.values()) & pool: + any_pooled = True + if expression: + source_id = self.traj_source_id(domain, target["trajectory_id"]) + rows = store._fts_rows(expression, 100000, source_ids=[source_id]) + if rows: + any_match = True + per_case[qid] = ( + "seeded_in_pool" if any_pooled + else "weak_seed_unpooled" if any_match + else "zero_seed" + ) + counts: dict[str, int] = {} + for value in per_case.values(): + counts[value] = counts.get(value, 0) + 1 + return {"per_case": per_case, "counts": counts} + + def case_pool_recovery(self, case: dict[str, Any], admitted: set[int]) -> dict[str, Any]: + """Did a NEW target state enter the (exact) pool via the expansion?""" + qid = case["qid"] + domain = case["domain"] + default_pool = self.default_pool_ids(qid) + result = {"recovered": False, "present_at_default": False, "new_target_states": 0} + if case["status"] == "pinned": + target_ids: set[int] = set() + for target in case["targets"]: + target_ids |= {int(v) for v in target["state_ids"].values()} + result["present_at_default"] = bool(target_ids & default_pool) + new = (target_ids & admitted) - default_pool + result["new_target_states"] = len(new) + result["recovered"] = bool(new) + elif case["status"] == "trajectory": + for target in case["targets"]: + states = set(self.traj_states(domain, target["trajectory_id"]).values()) + if states & default_pool: + result["present_at_default"] = True + continue # trajectory already reachable; not a recovery target + new = states & admitted + if new: + result["new_target_states"] += len(new) + result["recovered"] = True + return result + + def case_delivered_recall( + self, + case: dict[str, Any], + delivered: list[str], + baseline_delivered: list[str], + ) -> bool: + """Did a NEW target state reach the delivered 16-slot set?""" + got = _ref_pairs(delivered) + base = _ref_pairs(baseline_delivered) + if case["status"] == "pinned": + for target in case["targets"]: + for index in target["state_indices"]: + pair = (target["trajectory_id"], int(index)) + if pair in got and pair not in base: + return True + return False + if case["status"] == "trajectory": + base_trajs = {traj for traj, _idx in base} + for target in case["targets"]: + traj = target["trajectory_id"] + if traj in base_trajs: + continue + if any(t == traj for t, _idx in got): + return True + return False + return False + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--run-root", type=Path, default=_RUN_ROOT) + parser.add_argument("--h1-artifacts", type=Path, default=_H1) + parser.add_argument("--h31-artifacts", type=Path, default=_H31) + parser.add_argument("--targets", type=Path, default=_H5_TARGETS) + parser.add_argument("--out", type=Path, default=_H1 / "h5-replay-sweep.json") + parser.add_argument("--latency-sample", type=int, default=50) + parser.add_argument( + "--radii", type=int, nargs="+", default=[1, 2]) + parser.add_argument( + "--quotas", type=int, nargs="+", default=[8, 16, 32, 64, 128]) + args = parser.parse_args() + + ctx = ReplayContext(args.run_root, args.h1_artifacts, args.h31_artifacts) + h5 = H5Context(ctx, args.targets) + + # (iii) GOLDEN GATE -- mandatory before any sweep. + print("== GOLDEN GATE (defaults reproduce recorded delivery) ==", flush=True) + golden = golden_gate(ctx) + print(f" {golden['passed']}/{golden['total']} byte-identical; " + f"failures={golden['failures'][:10]}", flush=True) + if golden["passed"] != golden["total"]: + print("GOLDEN GATE FAILED -- aborting before sweep", flush=True) + return 1 + + # Seed split (knob-independent reachability boundary). + split = h5.seed_split() + print(f"\n== SEED SPLIT of the 30 == {split['counts']}", flush=True) + + # Preservation universe (same ground truth as the H3.1 instrument). + recon = json.loads( + (args.h1_artifacts / "old-rescore" / "reconciliation_sets.json").read_text() + ) + preserved_qids = sorted({item.split("/", 1)[1] for item in recon["preserved"]}) + + # Same-composition baselines (no adjacency). + print("\n== composition baselines (no adjacency) ==", flush=True) + target_qids = [case["qid"] for case in h5.cases] + loss_qids = [qid for _dom, qid in _LOSS_8] + baseline_delivered: dict[str, dict[str, list[str]]] = {} + baseline_latency: dict[str, dict[str, float]] = {} + sample = sorted(ctx.questions)[: args.latency_sample] + for comp_name, comp_kwargs in _COMPOSITIONS.items(): + per_qid: dict[str, list[str]] = {} + for qid in set(target_qids + loss_qids + preserved_qids): + per_qid[qid] = ctx.deliver(qid, **comp_kwargs) + baseline_delivered[comp_name] = per_qid + baseline_latency[comp_name] = measure_latency(ctx, sample, comp_kwargs) + print(f" {comp_name}: baseline p95 " + f"{baseline_latency[comp_name]['p95_ms']:.1f}ms", flush=True) + + knobs = [ + {"adjacency_radius": radius, "adjacency_quota": quota} + for radius in args.radii + for quota in args.quotas + ] + + results: list[dict[str, Any]] = [] + for knob in knobs: + label = f"r{knob['adjacency_radius']}q{knob['adjacency_quota']}" + started = time.perf_counter() + row: dict[str, Any] = {"knob": knob, "label": label} + + # Pool-entry recovery (composition-independent; expansion is + # pre-selection) + base-composition delivered recall from the SAME + # query call (base merged kwargs == the bare knob kwargs). + pool_cases: dict[str, dict[str, Any]] = {} + pool_recovered = 0 + by_bucket: dict[str, dict[str, int]] = {} + base_delivered_by_case: dict[str, list[str]] = {} + for case in h5.cases: + delivered, admitted = h5.deliver_and_admitted(case["qid"], **knob) + base_delivered_by_case[case["qid"]] = delivered + outcome = h5.case_pool_recovery(case, admitted) + pool_cases[case["qid"]] = outcome + slot = by_bucket.setdefault(case["bucket"], {"total": 0, "recovered": 0}) + slot["total"] += 1 + if outcome["recovered"]: + pool_recovered += 1 + slot["recovered"] += 1 + row["pool_recovery"] = pool_recovered + row["pool_recovery_by_bucket"] = by_bucket + row["pool_cases"] = pool_cases + + for comp_name, comp_kwargs in _COMPOSITIONS.items(): + merged = dict(comp_kwargs) + merged.update(knob) + base = baseline_delivered[comp_name] + delivered_recovered = 0 + delivered_case_ids: list[str] = [] + for case in h5.cases: + if comp_name == "base": + delivered = base_delivered_by_case[case["qid"]] + else: + delivered = ctx.deliver(case["qid"], **merged) + if h5.case_delivered_recall(case, delivered, base[case["qid"]]): + delivered_recovered += 1 + delivered_case_ids.append(case["qid"]) + loss_unchanged = 0 + loss_changed: list[str] = [] + for _dom, qid in _LOSS_8: + if ctx.deliver(qid, **merged) == base[qid]: + loss_unchanged += 1 + else: + loss_changed.append(qid) + disturbed: list[str] = [] + for qid in preserved_qids: + recorded = set(base[qid]) + now = set(ctx.deliver(qid, **merged)) + if recorded - now: + disturbed.append(qid) + # Paired latency read: the frozen-DB replay lives on an external + # drive with high I/O variance, so a knob pass is only comparable + # to a default pass measured back-to-back in the same phase. + default_latency = measure_latency(ctx, sample, comp_kwargs) + latency = measure_latency(ctx, sample, merged) + base_p95 = default_latency["p95_ms"] + row[comp_name] = { + "delivered_recall": delivered_recovered, + "delivered_recall_qids": delivered_case_ids, + "loss8_unchanged": loss_unchanged, + "loss8_changed_qids": loss_changed, + "preservation_disturbed": len(disturbed), + "preservation_disturbed_qids": disturbed, + "latency": latency, + "paired_default_latency": default_latency, + "latency_pct_vs_baseline": ( + (latency["p95_ms"] / base_p95 - 1.0) * 100.0 if base_p95 else 0.0 + ), + } + row["elapsed_s"] = round(time.perf_counter() - started, 1) + results.append(row) + print( + f" {label}: pool {pool_recovered}/30 | " + f"base dRec {row['base']['delivered_recall']}/30 " + f"loss8 {row['base']['loss8_unchanged']}/8 " + f"pres {row['base']['preservation_disturbed']}/{len(preserved_qids)} " + f"p95 {row['base']['latency_pct_vs_baseline']:+.1f}% | " + f"hyb dRec {row['hybrid']['delivered_recall']}/30 " + f"loss8 {row['hybrid']['loss8_unchanged']}/8 " + f"pres {row['hybrid']['preservation_disturbed']} " + f"p95 {row['hybrid']['latency_pct_vs_baseline']:+.1f}% " + f"({row['elapsed_s']}s)", + flush=True, + ) + + print("\n== SWEEP TABLE (component gate FROZEN; shipping knob NOT picked here) ==") + header = ( + f"{'knob':<8}{'poolRec/30':>11}" + f"{'b:dRec':>8}{'b:loss8':>9}{'b:pres':>8}{'b:p95Δ%':>9}" + f"{'h:dRec':>8}{'h:loss8':>9}{'h:pres':>8}{'h:p95Δ%':>9}" + ) + print(header) + print("-" * len(header)) + for row in results: + print( + f"{row['label']:<8}{row['pool_recovery']:>11}" + f"{row['base']['delivered_recall']:>8}" + f"{row['base']['loss8_unchanged']:>9}" + f"{row['base']['preservation_disturbed']:>8}" + f"{row['base']['latency_pct_vs_baseline']:>+8.1f}%" + f"{row['hybrid']['delivered_recall']:>8}" + f"{row['hybrid']['loss8_unchanged']:>9}" + f"{row['hybrid']['preservation_disturbed']:>8}" + f"{row['hybrid']['latency_pct_vs_baseline']:>+8.1f}%" + ) + + payload = { + "golden_gate": golden, + "seed_split": split, + "targets_manifest": str(args.targets), + "preserved_universe": len(preserved_qids), + "baseline_latency": baseline_latency, + "compositions": {k: {kk: list(vv) if isinstance(vv, tuple) else vv + for kk, vv in v.items()} + for k, v in _COMPOSITIONS.items()}, + "sweep": results, + } + args.out.write_text(json.dumps(payload, indent=2)) + print(f"\nwrote {args.out}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From da1f8f04390b15cdb217da844b942e7634a780d3 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 06:43:05 +0700 Subject: [PATCH 15/54] trajectory: per-state semantic index + backfill + default-off pool arm (#142) Add an additive per-STATE embedding space (the existing lcm_trajectory_embeddings is per-SOURCE / one coarse vector per trajectory, which cannot surface a lexically-invisible answer state). New idempotent tables lcm_trajectory_state_embedding{_profiles,s} created lazily via _ensure_state_semantic_schema (mirrors _ensure_semantic_schema; no existing-table change), a resumable build_state_semantic_index backfill (32-item/72K-token packing, chunked mean-pool path for over-cap states, skip-embedded resume, progress ledger callback), and a default-off state_semantic_quota query arm that admits the query's semantic nearest-neighbour states as a STRICTLY ADDITIVE _merge_arms tail (no boost, 5/traj cap intact, telemetry only when active). Defaults reproduce current bytes. 10 tests mirror the 8 adjacency tests + backfill resume/chunk/inert cases. --- ...est_trajectory_state_semantic_expansion.py | 447 +++++++++++++++ trajectory_store.py | 516 +++++++++++++++++- 2 files changed, 962 insertions(+), 1 deletion(-) create mode 100644 tests/test_trajectory_state_semantic_expansion.py diff --git a/tests/test_trajectory_state_semantic_expansion.py b/tests/test_trajectory_state_semantic_expansion.py new file mode 100644 index 000000000..04c5795ba --- /dev/null +++ b/tests/test_trajectory_state_semantic_expansion.py @@ -0,0 +1,447 @@ +"""State-level semantic pool-expansion + backfill (issue #142, Lane S / W3a). + +The per-SOURCE semantic index carries one coarse vector per trajectory and so +cannot surface a lexically-invisible answer STATE. These exercise the additive +per-STATE index that can: a target state carries NO query term of its own +(lexically invisible) but is the semantic nearest neighbour of the query, so the +``state_semantic_quota`` knob must pull it INTO the state pool as a quota-capped +additive tail -- while the defaults reproduce the pre-expansion pool and delivery +byte-for-byte, delivery stays unchanged when the ranked pool already fills the +nucleus (additive-only proof), the 5-per-trajectory diversity cap at selection is +preserved, and the backfill itself is resumable/idempotent (skip embedded +states) with a chunked path for over-cap documents. +""" + +from __future__ import annotations + +import hashlib +from pathlib import Path + +from hermes_lcm.trajectory_store import ( + CorpusIdentity, + TrajectorySource, + TrajectoryState, + TrajectoryStore, +) + + +class StateVectorProvider: + """A tiny 3-D embedder used to steer per-STATE ranking deterministically. + + A state whose text mentions ``alpha-answer`` embeds onto the +x axis (the + query direction), ``beta-answer`` onto +y, and everything else onto +z, so + the query (``embed_query`` -> +x) ranks exactly the alpha states first + regardless of their lexical (BM25) visibility. + """ + + provider_id = "fake" + model_id = "fake-state-v1" + + def __init__(self) -> None: + self.last_usage_tokens = 0 + self.document_calls = 0 + + @staticmethod + def _vector(text: str) -> list[float]: + folded = str(text).casefold() + if "alpha-answer" in folded: + return [1.0, 0.0, 0.0] + if "beta-answer" in folded: + return [0.0, 1.0, 0.0] + return [0.0, 0.0, 1.0] + + def embed_documents(self, texts): + self.document_calls += 1 + self.last_usage_tokens = sum(max(1, len(str(t)) // 4) for t in texts) + return [self._vector(t) for t in texts] + + def embed_query(self, text): # noqa: ARG002 + self.last_usage_tokens = 1 + return [1.0, 0.0, 0.0] + + +def _identity() -> CorpusIdentity: + return CorpusIdentity( + dataset_name="example/state-semantic", + dataset_revision="rev-state-semantic", + harness_commit="harness-state-semantic-1", + tier="small", + domain="web", + ingest_config_digest="state-semantic-test-v1", + ) + + +def _source(asset_root, *, trajectory_id, ordinal, goal, texts) -> TrajectorySource: + states = [] + for index, text in enumerate(texts): + screenshot = asset_root / f"{trajectory_id}-{index}.png" + screenshot.write_bytes(b"png" + hashlib.sha256(text.encode()).digest()) + states.append(TrajectoryState( + state_index=index, + step=index, + url=f"https://example.test/{trajectory_id}/{index}", + incoming_action=None if index == 0 else f"advance {index}", + thoughts=f"inspect state {index}", + text=text, + screenshot_path=screenshot, + )) + return TrajectorySource( + trajectory_id=trajectory_id, + ordinal=ordinal, + goal=goal, + start_url=f"https://example.test/{trajectory_id}", + outcome="completed", + states=tuple(states), + source_payload={"id": trajectory_id, "goal": goal}, + ) + + +_QUERY = "widget configuration export" + + +def _state_id(store, trajectory_id: str, state_index: int) -> int: + row = store._conn.execute( + """ + SELECT s.state_id FROM lcm_trajectory_states s + JOIN lcm_trajectory_sources src ON src.source_id = s.source_id + WHERE src.trajectory_id = ? AND s.state_index = ? + """, + (trajectory_id, state_index), + ).fetchone() + assert row is not None, f"unknown state {trajectory_id}/{state_index}" + return int(row[0]) + + +def _pool_state_ids(store) -> set[int]: + telemetry = store.last_query_telemetry() + return {int(item["state_id"]) for item in telemetry["state_candidate_pool"]} + + +def _admitted(store) -> list[dict[str, int]]: + telemetry = store.last_query_telemetry() + expansion = telemetry.get("state_semantic_expansion") + return list(expansion["admitted"]) if expansion else [] + + +def _build_invisible_semantic_store(tmp_path: Path, *, provider=None): + """One trajectory whose answer state is lexically INVISIBLE (no query term) + but is the semantic nearest neighbour (alpha-answer); plus a second lexical + trajectory so the query pool is non-empty.""" + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = provider or StateVectorProvider() + store = TrajectoryStore( + tmp_path / "lcm.db", + _identity(), + asset_root=asset_root, + embedding_provider=provider, + ) + store.insert(_source( + asset_root, + trajectory_id="answerpath", + ordinal=0, + goal="Update the account settings", + texts=( + "navigate homepage dashboard overview", # 0: invisible filler + "widget configuration export panel form", # 1: the lexical seed + "alpha-answer success banner shows view link", # 2: invisible ANSWER + "logout footer copyright notice", # 3: invisible filler + ), + )) + store.insert(_source( + asset_root, + trajectory_id="othertask", + ordinal=1, + goal="Export the report", + texts=("report export toolbar button",), + )) + store.finalize(["answerpath", "othertask"]) + store.build_state_semantic_index(provider) + return store + + +# --- default-off byte-identity ---------------------------------------------- + +def test_defaults_reproduce_current_bytes(tmp_path): + store = _build_invisible_semantic_store(tmp_path) + baseline = store.query(_QUERY, image_limit=0) + baseline_telemetry = store.last_query_telemetry() + explicit_off = store.query(_QUERY, image_limit=0, state_semantic_quota=0) + off_telemetry = store.last_query_telemetry() + assert [h.exact_ref for h in explicit_off] == [h.exact_ref for h in baseline] + assert off_telemetry == baseline_telemetry + # No telemetry key leaks into the default payload (frozen-run byte parity). + assert "state_semantic_expansion" not in baseline_telemetry + + +# --- the core recall mechanism ---------------------------------------------- + +def test_expansion_pulls_lexically_invisible_state_into_pool(tmp_path): + store = _build_invisible_semantic_store(tmp_path) + answer = _state_id(store, "answerpath", 2) + store.query(_QUERY, image_limit=0) + assert answer not in _pool_state_ids(store), "answer state must start invisible" + store.query(_QUERY, image_limit=0, state_semantic_quota=8) + assert answer in _pool_state_ids(store) + by_state = {entry["state_id"]: entry for entry in _admitted(store)} + assert answer in by_state + assert by_state[answer]["rank"] == 1 # the alpha state is the top-ranked + + +def test_quota_caps_admissions(tmp_path): + """Two lexically-invisible alpha states; a quota of 1 admits exactly one.""" + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = StateVectorProvider() + store = TrajectoryStore( + tmp_path / "lcm.db", _identity(), asset_root=asset_root, + embedding_provider=provider, + ) + store.insert(_source( + asset_root, + trajectory_id="answerpath", + ordinal=0, + goal="Update the account settings", + texts=( + "widget configuration export panel form", # lexical seed + "alpha-answer first invisible banner", # invisible alpha + "alpha-answer second invisible banner", # invisible alpha + ), + )) + store.finalize(["answerpath"]) + store.build_state_semantic_index(provider) + store.query(_QUERY, image_limit=0, state_semantic_quota=1) + admitted = _admitted(store) + assert len(admitted) == 1 + + +def test_pool_incumbents_are_not_readmitted(tmp_path): + """A state that already entered the pool lexically is never duplicated, even + when it is also the semantic top.""" + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = StateVectorProvider() + store = TrajectoryStore( + tmp_path / "lcm.db", _identity(), asset_root=asset_root, + embedding_provider=provider, + ) + store.insert(_source( + asset_root, + trajectory_id="allmatch", + ordinal=0, + goal="Finish the setup flow", + # Both matching states are ALSO alpha (lexical + semantic top); only the + # third is a non-matching invisible alpha the arm can add. + texts=( + "widget configuration export alpha-answer intro", + "widget configuration export alpha-answer detail", + "alpha-answer plain closing remark", + ), + )) + store.finalize(["allmatch"]) + store.build_state_semantic_index(provider) + store.query(_QUERY, image_limit=0, state_semantic_quota=8) + admitted = _admitted(store) + only = _state_id(store, "allmatch", 2) + assert [entry["state_id"] for entry in admitted] == [only] + pool = [ + int(item["state_id"]) + for item in store.last_query_telemetry()["state_candidate_pool"] + ] + assert len(pool) == len(set(pool)), "pool must stay duplicate-free" + + +# --- anti-filler / additive-only controls ----------------------------------- + +def test_delivery_unchanged_when_ranked_pool_fills_nucleus(tmp_path): + """Additive-only proof on a full pool: the semantic state may enter the POOL + but must not displace any delivered nucleus/backfill incumbent.""" + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = StateVectorProvider() + store = TrajectoryStore( + tmp_path / "lcm.db", _identity(), asset_root=asset_root, + embedding_provider=provider, + ) + order = [] + # Enough lexical trajectories to fill the delivered nucleus + backfill. + for index in range(12): + trajectory_id = f"lexical-{index:02d}" + store.insert(_source( + asset_root, + trajectory_id=trajectory_id, + ordinal=index, + goal="Configure the widget", + texts=(f"widget configuration export row {index}",), + )) + order.append(trajectory_id) + # One trajectory carrying the lexically-invisible semantic target. + store.insert(_source( + asset_root, + trajectory_id="target", + ordinal=12, + goal="Set up the dashboard gadget", + texts=( + "widget configuration export panel with all terms", + "alpha-answer invisible aftermath confirmation", + ), + )) + order.append("target") + store.finalize(order) + store.build_state_semantic_index(provider) + + baseline = [h.exact_ref for h in store.query(_QUERY, image_limit=0)] + expanded = [ + h.exact_ref + for h in store.query(_QUERY, image_limit=0, state_semantic_quota=16) + ] + assert expanded == baseline + admitted = _admitted(store) + assert admitted, "the pool itself must still gain a semantic entry" + invisible = _state_id(store, "target", 1) + assert invisible in {entry["state_id"] for entry in admitted} + + +def test_five_per_trajectory_cap_preserved_at_selection(tmp_path): + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = StateVectorProvider() + store = TrajectoryStore( + tmp_path / "lcm.db", _identity(), asset_root=asset_root, + embedding_provider=provider, + ) + # A long trajectory: one lexical seed + seven invisible alpha states, so the + # arm can admit MORE than five states from a single trajectory to the pool. + texts = ["widget configuration export summary step"] + texts += [f"alpha-answer invisible step {index}" for index in range(7)] + store.insert(_source( + asset_root, + trajectory_id="longtask", + ordinal=0, + goal="Complete the long procedure", + texts=tuple(texts), + )) + store.finalize(["longtask"]) + store.build_state_semantic_index(provider) + hits = store.query( + _QUERY, + image_limit=0, + include_adjacent=False, # nucleus only: isolate the selection cap + state_semantic_quota=32, + ) + per_trajectory: dict[str, int] = {} + for hit in hits: + per_trajectory[hit.trajectory_id] = per_trajectory.get(hit.trajectory_id, 0) + 1 + assert per_trajectory.get("longtask", 0) <= 5 + # ...even though MORE than five longtask states were admitted to the pool. + assert len(_admitted(store)) == 7 + + +def test_expanded_states_selected_only_from_the_tail(tmp_path): + """Semantic-admitted states earn no BM25 rank: every ranked pool row still + precedes every admitted state-semantic row in the candidate pool.""" + store = _build_invisible_semantic_store(tmp_path) + store.query(_QUERY, image_limit=0, state_semantic_quota=8) + telemetry = store.last_query_telemetry() + pool = [int(item["state_id"]) for item in telemetry["state_candidate_pool"]] + admitted_ids = {entry["state_id"] for entry in _admitted(store)} + ranked_positions = [ + index for index, state_id in enumerate(pool) if state_id not in admitted_ids + ] + admitted_positions = [ + index for index, state_id in enumerate(pool) if state_id in admitted_ids + ] + assert admitted_positions, "admitted states must be visible in the pool" + assert max(ranked_positions) < min(admitted_positions) + + +# --- backfill: resumability + inert-without-index --------------------------- + +def test_backfill_is_idempotent_and_resumable(tmp_path): + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = StateVectorProvider() + store = TrajectoryStore( + tmp_path / "lcm.db", _identity(), asset_root=asset_root, + embedding_provider=provider, + ) + store.insert(_source( + asset_root, + trajectory_id="answerpath", + ordinal=0, + goal="Update the account settings", + texts=( + "widget configuration export panel form", + "alpha-answer success banner", + "logout footer copyright notice", + ), + )) + store.finalize(["answerpath"]) + first = store.build_state_semantic_index(provider) + assert first["states_embedded"] == 3 + assert first["total_states"] == 3 + assert first["provider_calls"] >= 1 + # A re-run embeds nothing (every state already carries a row). + provider.document_calls = 0 + second = store.build_state_semantic_index(provider) + assert second["states_embedded"] == 0 + assert second["already_embedded"] == 3 + assert provider.document_calls == 0 + + +def test_chunked_path_pools_oversize_documents(tmp_path): + """A document over the (test-lowered) per-document token budget takes the + chunked path and still yields exactly one usable per-state vector.""" + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = StateVectorProvider() + store = TrajectoryStore( + tmp_path / "lcm.db", _identity(), asset_root=asset_root, + embedding_provider=provider, + ) + long_text = "alpha-answer " + ("token " * 400) # well over a 5-token budget + store.insert(_source( + asset_root, + trajectory_id="answerpath", + ordinal=0, + goal="Update the account settings", + texts=( + "widget configuration export panel form", + long_text, + ), + )) + store.finalize(["answerpath"]) + stats = store.build_state_semantic_index( + provider, document_token_budget=5, batch_token_budget=50, batch_max_items=4 + ) + assert stats["chunked_states"] == 1 + assert stats["states_embedded"] == 2 + oversize = _state_id(store, "answerpath", 1) + row = store._conn.execute( + "SELECT vector FROM lcm_trajectory_state_embeddings WHERE state_id = ?", + (oversize,), + ).fetchone() + assert row is not None and len(bytes(row["vector"])) == stats["dim"] * 4 + + +def test_arm_inert_without_provider_or_index(tmp_path): + """Knob-on but no state index (or no provider) is a no-op, not an error.""" + asset_root = tmp_path / "assets" + asset_root.mkdir() + # No provider attached and no state backfill performed. + store = TrajectoryStore(tmp_path / "lcm.db", _identity(), asset_root=asset_root) + store.insert(_source( + asset_root, + trajectory_id="answerpath", + ordinal=0, + goal="Update the account settings", + texts=("widget configuration export panel form", "plain closing remark"), + )) + store.finalize(["answerpath"]) + baseline = [h.exact_ref for h in store.query(_QUERY, image_limit=0)] + with_knob = [ + h.exact_ref + for h in store.query(_QUERY, image_limit=0, state_semantic_quota=8) + ] + assert with_knob == baseline + assert _admitted(store) == [] diff --git a/trajectory_store.py b/trajectory_store.py index 56ba517a8..f233c93ad 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -19,7 +19,7 @@ import struct import threading import time -from typing import Any, Iterable, Protocol, Sequence +from typing import Any, Callable, Iterable, Protocol, Sequence from urllib.parse import quote, unquote from .db_bootstrap import ( @@ -35,6 +35,19 @@ TRAJECTORY_MIGRATION_STEP = "trajectory_store_v1" TRAJECTORY_SCHEMA_VERSION = 1 TRAJECTORY_SEMANTIC_DOCUMENT_VERSION = "trajectory-semantic-document-v1" +# Per-STATE semantic index (issue #142, Lane S / W3a). A separate, additive +# embedding space keyed by state_id (the source-level index above is one coarse +# vector per trajectory). Bumping this string invalidates a state backfill the +# same way the source version does. +TRAJECTORY_STATE_SEMANTIC_DOCUMENT_VERSION = "trajectory-state-semantic-document-v1" +# Voyage single-request caps (mirrors embedding_provider's budgets, applied with +# the same 0.9 safety factor the provider uses): one document <= 27K tokens, one +# request <= 80K tokens. The state backfill packs 32 items / 72K tokens per +# request (matches the #141 sizing simulation); a state above the per-document +# budget takes the chunked path (token-window split, mean-pooled). +_STATE_EMBED_MAX_BATCH_ITEMS = 32 +_STATE_EMBED_DOCUMENT_TOKEN_BUDGET = int(27_000 * 0.9) +_STATE_EMBED_BATCH_TOKEN_BUDGET = int(80_000 * 0.9) _MAX_CANDIDATES = 128 _MAX_RESULTS = 24 _MAX_IMAGES = 8 @@ -402,6 +415,12 @@ def __init__( } self._last_semantic_attempt: TrajectorySemanticAttempt | None = None self._last_query_telemetry: dict[str, Any] | None = None + # Lazily-populated per-STATE semantic matrix cache (issue #142): the + # tuple is ``(profile_digest, state_ids, matrix)`` where ``matrix`` is a + # normalized float32 array (numpy when available, else a list of tuples) + # so query-time state ranking is a single mat-vec instead of a per-row + # SQL + Python dot loop. Reset whenever a state backfill rewrites rows. + self._state_semantic_cache: tuple[str, list[int], Any] | None = None self._lock = threading.RLock() self._conn = self._open_connection() try: @@ -1277,6 +1296,368 @@ def build_semantic_index( def semantic_metrics(self) -> dict[str, int]: return dict(self._semantic_usage) + # ------------------------------------------------------------------ + # Per-STATE semantic index (issue #142, Lane S / W3a). + # + # An ADDITIVE second embedding space keyed by ``state_id`` -- the coarse + # per-trajectory index above cannot surface a lexically-invisible answer + # state, so the recall lane needs a per-state vector. The tables are created + # lazily/idempotently exactly like ``_ensure_semantic_schema`` (no change to + # any existing table); the backfill is resumable (skip embedded states) and + # drives its own request packing so a bulk run stays inside Voyage's caps. + # ------------------------------------------------------------------ + + def _ensure_state_semantic_schema(self) -> None: + self._require_writable() + self._conn.executescript( + """ + CREATE TABLE IF NOT EXISTS lcm_trajectory_state_embedding_profiles ( + profile_digest TEXT PRIMARY KEY, + provider TEXT NOT NULL, + model_name TEXT NOT NULL, + dim INTEGER NOT NULL CHECK(dim > 0), + document_version TEXT NOT NULL, + source_manifest_digest TEXT NOT NULL, + state_count INTEGER NOT NULL CHECK(state_count >= 0), + active INTEGER NOT NULL DEFAULT 0 CHECK(active IN (0, 1)), + created_at REAL NOT NULL + ); + + CREATE UNIQUE INDEX IF NOT EXISTS lcm_trajectory_state_embedding_one_active + ON lcm_trajectory_state_embedding_profiles(active) WHERE active = 1; + + CREATE TABLE IF NOT EXISTS lcm_trajectory_state_embeddings ( + state_id INTEGER PRIMARY KEY + REFERENCES lcm_trajectory_states(state_id) ON DELETE CASCADE, + profile_digest TEXT NOT NULL + REFERENCES lcm_trajectory_state_embedding_profiles(profile_digest) + ON DELETE CASCADE, + document_sha256 TEXT NOT NULL, + vector BLOB NOT NULL, + embedded_at REAL NOT NULL + ); + + CREATE INDEX IF NOT EXISTS lcm_trajectory_state_embeddings_profile + ON lcm_trajectory_state_embeddings(profile_digest, state_id); + """ + ) + self._conn.commit() + + def _state_semantic_profile_exists(self) -> bool: + return self._conn.execute( + """ + SELECT 1 FROM sqlite_master + WHERE type = 'table' AND name = 'lcm_trajectory_state_embedding_profiles' + """ + ).fetchone() is not None + + def active_state_semantic_profile(self) -> sqlite3.Row | None: + if not self._state_semantic_profile_exists(): + return None + return self._conn.execute( + "SELECT * FROM lcm_trajectory_state_embedding_profiles WHERE active = 1" + ).fetchone() + + @staticmethod + def _state_semantic_profile_digest( + provider_name: str, model_name: str, dim: int, source_manifest_digest: str + ) -> str: + return _sha256_text(_canonical_json({ + "provider": provider_name, + "model": model_name, + "dim": int(dim), + "document_version": TRAJECTORY_STATE_SEMANTIC_DOCUMENT_VERSION, + "source_manifest_digest": source_manifest_digest, + })) + + @staticmethod + def _state_embed_document(text: str, url: str, state_id: int) -> str: + """The per-state embedding document. ``states.text`` is the FTS-indexed + visible content (what the #141 sizing counted); an empty text falls back + to the URL, then a stable state marker, so the provider never receives an + empty input (Voyage rejects those).""" + candidate = str(text or "").strip() + if candidate: + return str(text) + candidate = str(url or "").strip() + if candidate: + return str(url) + return f"state-{int(state_id)}" + + def _state_token_chunks(self, document: str, token_budget: int) -> list[str]: + """Split ``document`` into contiguous pieces each <= ``token_budget`` + tokens (the chunked path for the ~228 states over Voyage's per-document + cap). Uses the shared cl100k encoder -- the same tokenizer the provider's + packing uses to gate the caps -- and falls back to a conservative + character window if the encoder is unavailable.""" + from .tokens import _get_encoder + + encoder = _get_encoder() + if encoder is None: + # ~4 chars/token is the module's own char fallback; stay under budget. + span = max(1, token_budget * 4) + return [document[i:i + span] for i in range(0, len(document), span)] + token_ids = encoder.encode(document) + chunks: list[str] = [] + for start in range(0, len(token_ids), token_budget): + piece = encoder.decode(token_ids[start:start + token_budget]) + if piece: + chunks.append(piece) + return chunks or [document] + + def build_state_semantic_index( + self, + provider: TrajectoryEmbeddingProvider | None = None, + *, + resume: bool = True, + batch_max_items: int = _STATE_EMBED_MAX_BATCH_ITEMS, + batch_token_budget: int = _STATE_EMBED_BATCH_TOKEN_BUDGET, + document_token_budget: int = _STATE_EMBED_DOCUMENT_TOKEN_BUDGET, + progress_callback: Callable[[dict[str, Any]], None] | None = None, + ) -> dict[str, Any]: + """Resumable per-state embedding backfill (issue #142). + + Embeds one vector per state (``states.text``) into + ``lcm_trajectory_state_embeddings`` under an active profile keyed by + provider/model/dim/document_version/source_manifest_digest. Requests are + packed to ``batch_max_items`` items / ``batch_token_budget`` tokens; a + state whose document exceeds ``document_token_budget`` is split into + token windows, each embedded, and the pieces mean-pooled + normalized + into a single state vector (same billed tokens, more requests). + + Resumable + idempotent: with ``resume=True`` (default) states that + already carry a row under the active profile are skipped, so a re-run + after an interruption embeds only the remainder and a fully-embedded + store makes ZERO provider calls. ``progress_callback`` is invoked after + each dispatched request with a cumulative-stats dict (for a live ledger). + """ + from .tokens import count_tokens + + if self.status != "complete": + raise CorpusIdentityError( + "trajectory corpus must be finalized before state indexing" + ) + active_provider = provider or self.embedding_provider + if active_provider is None: + raise TrajectoryStoreError("trajectory embedding provider is not configured") + self._ensure_state_semantic_schema() + corpus = self._conn.execute( + """ + SELECT source_manifest_digest, trajectory_count + FROM lcm_trajectory_corpora WHERE singleton = 1 + """ + ).fetchone() + source_manifest_digest = str(corpus["source_manifest_digest"]) + provider_name = str(getattr(active_provider, "provider_id", "unknown")) + model_name = str(getattr(active_provider, "model_id", "")).strip() + if not model_name: + raise ValueError("trajectory embedding model_id must not be empty") + + batch_max_items = max(1, int(batch_max_items)) + batch_token_budget = max(1, int(batch_token_budget)) + document_token_budget = max(1, int(document_token_budget)) + + # Discover the embedding dimension without a wasted call when possible: + # the source-level active profile shares this provider/model, so its + # dim is authoritative; otherwise probe one state. + source_profile = self._semantic_profile() + dim: int | None = None + if ( + source_profile is not None + and str(source_profile["provider"]) == provider_name + and str(source_profile["model_name"]) == model_name + ): + dim = int(source_profile["dim"]) + + all_states = self._conn.execute( + """ + SELECT s.state_id, s.text, s.url + FROM lcm_trajectory_states s + ORDER BY s.state_id + """ + ).fetchall() + total_states = len(all_states) + + if dim is None: + if not all_states: + raise ValueError("trajectory corpus has no states to index") + probe_doc = self._state_embed_document( + all_states[0]["text"], all_states[0]["url"], int(all_states[0]["state_id"]) + ) + probe_vector = _normalized_vector(active_provider.embed_query(probe_doc)) + dim = len(probe_vector) + + profile_digest = self._state_semantic_profile_digest( + provider_name, model_name, dim, source_manifest_digest + ) + now = time.time() + with self._lock: + self._conn.execute("BEGIN IMMEDIATE") + try: + self._conn.execute( + "UPDATE lcm_trajectory_state_embedding_profiles " + "SET active = 0 WHERE active = 1 AND profile_digest != ?", + (profile_digest,), + ) + self._conn.execute( + """ + INSERT INTO lcm_trajectory_state_embedding_profiles( + profile_digest, provider, model_name, dim, + document_version, source_manifest_digest, + state_count, active, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, 1, ?) + ON CONFLICT(profile_digest) DO UPDATE SET + state_count = excluded.state_count, + active = 1 + """, + ( + profile_digest, + provider_name, + model_name, + int(dim), + TRAJECTORY_STATE_SEMANTIC_DOCUMENT_VERSION, + source_manifest_digest, + total_states, + now, + ), + ) + self._conn.commit() + except Exception: + self._conn.rollback() + raise + + already: set[int] = set() + if resume: + already = { + int(row[0]) + for row in self._conn.execute( + "SELECT state_id FROM lcm_trajectory_state_embeddings " + "WHERE profile_digest = ?", + (profile_digest,), + ) + } + pending = [row for row in all_states if int(row["state_id"]) not in already] + + stats: dict[str, Any] = { + "profile_digest": profile_digest, + "dim": int(dim), + "total_states": total_states, + "already_embedded": len(already), + "pending": len(pending), + "states_embedded": 0, + "chunked_states": 0, + "provider_calls": 0, + "billed_tokens": 0, + } + + # Partition pending states into single-request documents and the + # oversize (chunked) minority, both packed to the same item/token caps. + normal: list[tuple[int, str, str, int]] = [] # (state_id, doc, sha, tokens) + oversize: list[tuple[int, str, str]] = [] # (state_id, doc, sha) + for row in pending: + state_id = int(row["state_id"]) + document = self._state_embed_document(row["text"], row["url"], state_id) + document_sha = _sha256_text(document) + tokens = count_tokens(document) + if tokens > document_token_budget: + oversize.append((state_id, document, document_sha)) + else: + normal.append((state_id, document, document_sha, tokens)) + + def _persist(rows: list[tuple[int, str, bytes]]) -> None: + with self._lock: + self._conn.execute("BEGIN IMMEDIATE") + try: + self._conn.executemany( + """ + INSERT INTO lcm_trajectory_state_embeddings( + state_id, profile_digest, document_sha256, vector, embedded_at + ) VALUES (?, ?, ?, ?, ?) + ON CONFLICT(state_id) DO UPDATE SET + profile_digest = excluded.profile_digest, + document_sha256 = excluded.document_sha256, + vector = excluded.vector, + embedded_at = excluded.embedded_at + """, + [ + (sid, profile_digest, sha, vec, time.time()) + for sid, sha, vec in rows + ], + ) + self._conn.commit() + except Exception: + self._conn.rollback() + raise + + def _emit_progress() -> None: + if progress_callback is not None: + progress_callback(dict(stats)) + + # --- normal single-request states ------------------------------------- + batch_ids: list[int] = [] + batch_docs: list[str] = [] + batch_shas: list[str] = [] + batch_tokens = 0 + + def _flush_normal() -> None: + nonlocal batch_ids, batch_docs, batch_shas, batch_tokens + if not batch_docs: + return + vectors = active_provider.embed_documents(batch_docs) + if len(vectors) != len(batch_docs): + raise ValueError("state embedding count does not match batch size") + packed = [] + for sid, sha, vector in zip(batch_ids, batch_shas, vectors): + normalized = _normalized_vector(vector, expected_dim=dim) + packed.append((sid, sha, _pack_vector(normalized))) + _persist(packed) + stats["provider_calls"] += 1 + stats["billed_tokens"] += max( + 0, int(getattr(active_provider, "last_usage_tokens", 0) or 0) + ) + stats["states_embedded"] += len(packed) + batch_ids, batch_docs, batch_shas, batch_tokens = [], [], [], 0 + _emit_progress() + + for state_id, document, document_sha, tokens in normal: + if batch_docs and ( + len(batch_docs) >= batch_max_items + or batch_tokens + tokens > batch_token_budget + ): + _flush_normal() + batch_ids.append(state_id) + batch_docs.append(document) + batch_shas.append(document_sha) + batch_tokens += tokens + _flush_normal() + + # --- oversize (chunked) states ---------------------------------------- + for state_id, document, document_sha in oversize: + chunks = self._state_token_chunks(document, document_token_budget) + chunk_vectors: list[tuple[float, ...]] = [] + for start in range(0, len(chunks), batch_max_items): + sub = chunks[start:start + batch_max_items] + vectors = active_provider.embed_documents(sub) + if len(vectors) != len(sub): + raise ValueError("chunk embedding count does not match batch size") + for vector in vectors: + chunk_vectors.append(_normalized_vector(vector, expected_dim=dim)) + stats["provider_calls"] += 1 + stats["billed_tokens"] += max( + 0, int(getattr(active_provider, "last_usage_tokens", 0) or 0) + ) + pooled = [sum(values) / len(chunk_vectors) for values in zip(*chunk_vectors)] + normalized = _normalized_vector(pooled, expected_dim=dim) + _persist([(state_id, document_sha, _pack_vector(normalized))]) + stats["states_embedded"] += 1 + stats["chunked_states"] += 1 + _emit_progress() + + # A rewrite invalidates any cached query-time matrix. + self._state_semantic_cache = None + stats["status"] = "current" if not pending else "built" + return stats + @staticmethod def _safe_getattr(obj: Any, name: str) -> Any: """getattr that swallows EVERY exception, not just AttributeError. @@ -1450,6 +1831,86 @@ def _semantic_source_ranks(self, query: str) -> list[tuple[int, float]]: ranked.sort(key=lambda item: (-item[1], item[0])) return ranked[: self.semantic_top_trajectories] + def _load_state_semantic_matrix( + self, profile_digest: str, dim: int + ) -> tuple[list[int], Any]: + """Cache and return ``(state_ids, matrix)`` for the active state profile. + + ``matrix`` is a normalized ``float32`` numpy array when numpy is present + (a single mat-vec ranks all states) and otherwise a list of vector + tuples for a pure-Python fallback -- the state vectors were normalized at + backfill, so a query-vector dot product is cosine similarity either way. + """ + cache = self._state_semantic_cache + if cache is not None and cache[0] == profile_digest: + return cache[1], cache[2] + rows = self._conn.execute( + """ + SELECT state_id, vector FROM lcm_trajectory_state_embeddings + WHERE profile_digest = ? + ORDER BY state_id + """, + (profile_digest,), + ).fetchall() + state_ids = [int(row["state_id"]) for row in rows] + try: + import numpy as _np + + if rows: + matrix: Any = _np.frombuffer( + b"".join(bytes(row["vector"]) for row in rows), dtype=" list[tuple[int, float]]: + """Top-``top_k`` ``(state_id, similarity)`` for the query against the + active per-state semantic index (issue #142). Returns ``[]`` when no + provider or no active state profile is present (the arm is then inert).""" + provider = self.embedding_provider + profile = self.active_state_semantic_profile() + if provider is None or profile is None or top_k <= 0: + return [] + if ( + str(profile["provider"]) != str(getattr(provider, "provider_id", "unknown")) + or str(profile["model_name"]) != str(getattr(provider, "model_id", "")) + ): + return [] + dim = int(profile["dim"]) + state_ids, matrix = self._load_state_semantic_matrix( + str(profile["profile_digest"]), dim + ) + if not state_ids: + return [] + query_vector = _normalized_vector(provider.embed_query(query), expected_dim=dim) + try: + import numpy as _np + + query_array = _np.asarray(query_vector, dtype=" str: terms: list[str] = [] @@ -1820,6 +2281,7 @@ def query( arm_quota: tuple[int, int] | None = None, adjacency_radius: int = 0, adjacency_quota: int = 0, + state_semantic_quota: int = 0, ) -> tuple[TrajectoryHit, ...]: if self.status != "complete": raise CorpusIdentityError("trajectory corpus must be finalized before query") @@ -1829,6 +2291,7 @@ def query( lexical_floor = min(max(0, int(lexical_floor)), _MAX_RESULTS) adjacency_radius = min(max(0, int(adjacency_radius)), _MAX_ADJACENCY_RADIUS) adjacency_quota = min(max(0, int(adjacency_quota)), _MAX_CANDIDATES) + state_semantic_quota = min(max(0, int(state_semantic_quota)), _MAX_CANDIDATES) text_char_limit = min( max(256, int(text_char_limit)), _MAX_QUERY_TEXT_CHARS, @@ -1976,6 +2439,50 @@ def query( }) rows = expanded + # State-level semantic pool-expansion (issue #142, Lane S / W3a): rank + # the per-state semantic index against the query and admit up to + # ``state_semantic_quota`` states that are NOT already in the pool, as a + # STRICTLY ADDITIVE tail through the same ``_merge_arms`` machinery as the + # adjacency arm -- no semantic boost, no BM25 rank of their own, appended + # AFTER the ranked pool so the nucleus (and delivery) only changes when + # the ranked pool alone underfills. ``state_semantic_quota == 0`` + # (default), no provider, or no active state index skip this entirely and + # reproduce current bytes. Independent of the adjacency arm above. + state_semantic_admitted: list[dict[str, Any]] = [] + if state_semantic_quota > 0 and rows: + pool_ids = {int(row["state_id"]) for row in rows} + ranked_states = self._semantic_state_ranks( + query, state_semantic_quota + len(pool_ids) + 16 + ) + score_by_state = {sid: score for sid, score in ranked_states} + arm_semantic = [ + {"state_id": sid} + for sid, _score in ranked_states + if sid not in pool_ids + ] + if arm_semantic: + merged = self._merge_arms( + rows, + arm_semantic, # type: ignore[arg-type] # only "state_id" is read + len(rows) + state_semantic_quota, + len(rows), + state_semantic_quota, + ) + admitted_ids = [int(row["state_id"]) for row in merged[len(rows):]] + full_by_id = self._state_rows_by_ids(admitted_ids) + expanded = list(rows) + for rank, state_id in enumerate(admitted_ids, start=1): + similarity = float(score_by_state.get(state_id, 0.0)) + expanded.append(full_by_id[state_id]) + candidate_kind[state_id] = "state_semantic" + candidate_score[state_id] = similarity + state_semantic_admitted.append({ + "state_id": state_id, + "rank": rank, + "score": similarity, + }) + rows = expanded + adjacent_reserve = min(6, limit // 3) if include_adjacent else 0 nucleus_limit = max(1, limit - adjacent_reserve) if arm_quota is not None: @@ -2097,6 +2604,13 @@ def query( "quota": adjacency_quota, "admitted": adjacency_admitted, } + if state_semantic_quota > 0: + # Present only when the state-semantic knob is active so the default + # telemetry payload stays byte-identical (golden 451/451). + self._last_query_telemetry["state_semantic_expansion"] = { + "quota": state_semantic_quota, + "admitted": state_semantic_admitted, + } return tuple(hits) def resolve_exact_ref(self, exact_ref: str) -> TrajectoryHit: From 1df8821075da2bcb34a96f74a5f858e83125148d Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 07:46:25 +0700 Subject: [PATCH 16/54] bench: state-embedding backfill CLI + state-semantic replay sweep (#142) state_embedding_backfill.py: resumable metered CLI over a working-copy lcm.db (32-item/72K packing via build_state_semantic_index, JSONL spend ledger, projected-cost cost-cap abort, 50%% checkpoint). h5_state_semantic_replay.py: sibling of the adjacency sweep -- opens the BACKFILLED copies with a cached real Voyage query provider (source ranks still injected -> golden byte-identical), sweeps state_semantic_quota and reports the four frozen #142 gate metrics (pool-entry/30, delivered-recall@16, loss8 preservation, 154-set preservation, p95). Query embeds cached off the timed path so the latency delta isolates the arm's added cost. --- benchmarking/h5_state_semantic_replay.py | 283 +++++++++++++++++++++++ benchmarking/state_embedding_backfill.py | 219 ++++++++++++++++++ 2 files changed, 502 insertions(+) create mode 100644 benchmarking/h5_state_semantic_replay.py create mode 100644 benchmarking/state_embedding_backfill.py diff --git a/benchmarking/h5_state_semantic_replay.py b/benchmarking/h5_state_semantic_replay.py new file mode 100644 index 000000000..c12f7ca4f --- /dev/null +++ b/benchmarking/h5_state_semantic_replay.py @@ -0,0 +1,283 @@ +#!/usr/bin/env python3 +"""Replay sweep for the state-level semantic pool-expansion (issue #142, W3a). + +Sibling of ``h5_recall_replay`` (the H5(b) adjacency sweep). Same frozen assets, +same injected SOURCE ranks (so the base fused pool is byte-identical), same +golden gate and the same four frozen gate metrics on the same target/preservation +sets -- but the arm under test is ``state_semantic_quota`` against a REAL +per-state Voyage index that has been backfilled into the store copies. + +Key differences from the provider-free adjacency sweep: + * Stores are opened over the BACKFILLED working copies (which carry the + ``lcm_trajectory_state_embeddings`` table), NOT the bare h31 db-copies. + * A real Voyage QUERY provider is attached so the state arm can embed the + query and rank the per-state vectors. Query embeddings are CACHED (embedded + once, off the timed path) exactly as source ranks are injected -- so the + latency gate isolates the arm's added algorithmic cost (mat-vec + a small + SQL fetch), which is what production pays on top of the query embed the + source-semantic path already performs. Source ranks stay INJECTED so the + golden gate is byte-identical. + +Reported per knob (component gate FROZEN; the orchestrator applies it): + POOL-ENTRY RECOVERY -- of the 30 h5-targets, how many gain a NEW target state + in the pool via the admitted state-semantic tail. + DELIVERED-RECALL@16 -- new target state reaches the delivered top-16. + ANTI-FILLER (loss8) -- the 8 H32 loss-ids' delivered sets unchanged. + PRESERVATION -- stable-correct (154-set) delivered refs dropped. + LATENCY p95 -- vs the same-composition no-arm pass (cached embeds). +""" +from __future__ import annotations + +import argparse +import json +import os +import sqlite3 +import sys +import time +from pathlib import Path +from typing import Any + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from benchmarking.h3_composition_replay import ( # noqa: E402 + _H1, _H31, _RUN_ROOT, ReplayContext, golden_gate, measure_latency, +) +from benchmarking.h5_recall_replay import ( # noqa: E402 + _LOSS_8, H5Context, +) + +_H5_TARGETS = _H1 / "h5-targets.json" +_DEFAULT_DB_DIR = Path("/Volumes/LEXAR/hermes-work/W3A-dbwork") + + +class CachedVoyageQueryProvider: + """Real Voyage query embedding, cached by text so repeated timed passes make + zero network calls (the source ranks are injected the same way). Reports the + voyage/voyage-4 identity the backfilled state profile was written under.""" + + provider_id = "voyage" + model_id = "voyage-4" + + def __init__(self, model: str = "voyage-4", timeout: float = 30.0) -> None: + from hermes_lcm.embedding_provider import VoyageProvider + + self._real = VoyageProvider(model, timeout=timeout) + self.model_id = model + self._cache: dict[str, list[float]] = {} + self.last_usage_tokens = 0 + self.calls = 0 + + def embed_query(self, text: str) -> list[float]: + cached = self._cache.get(text) + if cached is None: + self.calls += 1 + cached = list(self._real.embed_query(text)) + self._cache[text] = cached + return cached + + def embed_documents(self, texts): # noqa: ARG002 + raise RuntimeError("CachedVoyageQueryProvider is query-only") + + +class StateReplayContext(ReplayContext): + """ReplayContext whose stores are the BACKFILLED copies with a real (cached) + Voyage query provider attached, so the state-semantic arm is live while the + injected source ranks keep the base pool byte-identical.""" + + def __init__(self, run_root, h1, h31, db_dir: Path, provider) -> None: + self._db_dir = Path(db_dir) + self._provider = provider + super().__init__(run_root, h1, h31) + + def _open_store(self, db_path, domain): # noqa: ARG002 + real_db = self._db_dir / f"{domain}.lcm.db" + conn = sqlite3.connect(f"file:{real_db}?mode=ro", uri=True) + identity_json = json.loads( + conn.execute( + "SELECT identity_json FROM lcm_trajectory_corpora WHERE singleton=1" + ).fetchone()[0] + ) + conn.close() + identity = self._ts.CorpusIdentity( + dataset_name=identity_json["dataset_name"], + dataset_revision=identity_json["dataset_revision"], + harness_commit=identity_json["harness_commit"], + tier=identity_json["tier"], + domain=identity_json["domain"], + ingest_config_digest=identity_json.get("ingest_config_digest", ""), + ) + base = self._ts.TrajectoryStore + + class _ReplayStore(base): # type: ignore[misc, valid-type] + injected: list[tuple[int, float]] = [] + + def _semantic_source_ranks(self, query: str): # noqa: ARG002 + return list(self.injected) + + store = _ReplayStore( + real_db, identity, asset_root=real_db.parent, + read_only=True, semantic_top_trajectories=12, + ) + store.embedding_provider = self._provider + return store + + +def _state_admitted(ctx: ReplayContext, qid: str) -> set[int]: + domain, _ = ctx.questions[qid] + telemetry = ctx.stores[domain].last_query_telemetry() or {} + expansion = telemetry.get("state_semantic_expansion") or {} + return {int(e["state_id"]) for e in expansion.get("admitted", [])} + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--run-root", type=Path, default=_RUN_ROOT) + parser.add_argument("--h1-artifacts", type=Path, default=_H1) + parser.add_argument("--h31-artifacts", type=Path, default=_H31) + parser.add_argument("--targets", type=Path, default=_H5_TARGETS) + parser.add_argument("--db-dir", type=Path, default=_DEFAULT_DB_DIR) + parser.add_argument("--out", type=Path, default=_H1 / "W3A-state-sweep.json") + parser.add_argument("--latency-sample", type=int, default=50) + parser.add_argument("--quotas", type=int, nargs="+", default=[4, 8, 16]) + args = parser.parse_args() + + if not os.environ.get("VOYAGE_API_KEY", "").strip(): + print("VOYAGE_API_KEY is not set -- the state arm needs a query embedder", + flush=True) + return 3 + + provider = CachedVoyageQueryProvider() + ctx = StateReplayContext( + args.run_root, args.h1_artifacts, args.h31_artifacts, args.db_dir, provider + ) + h5 = H5Context(ctx, args.targets) + + # (iii) GOLDEN GATE -- knob-off must reproduce recorded delivery byte-for-byte. + print("== GOLDEN GATE (defaults-off reproduce recorded delivery) ==", flush=True) + golden = golden_gate(ctx) + print(f" {golden['passed']}/{golden['total']} byte-identical; " + f"failures={golden['failures'][:10]}", flush=True) + if golden["passed"] != golden["total"]: + print("GOLDEN GATE FAILED -- aborting before sweep", flush=True) + return 1 + + split = h5.seed_split() + print(f"\n== SEED SPLIT of the 30 == {split['counts']}", flush=True) + + recon = json.loads( + (args.h1_artifacts / "old-rescore" / "reconciliation_sets.json").read_text() + ) + preserved_qids = sorted({item.split("/", 1)[1] for item in recon["preserved"]}) + + target_qids = [case["qid"] for case in h5.cases] + loss_qids = [qid for _dom, qid in _LOSS_8] + all_qids = sorted(set(target_qids + loss_qids + preserved_qids)) + sample = sorted(ctx.questions)[: args.latency_sample] + + # WARM the query-vector cache + the state matrix so timed passes are network + # -free (mirrors the injected source ranks). One pass at the largest quota. + warm_quota = max(args.quotas) + print(f"\n== warming query-embed cache (quota={warm_quota}) ==", flush=True) + for qid in sorted(set(all_qids + sample)): + ctx.deliver(qid, state_semantic_quota=warm_quota) + print(f" cached {provider.calls} unique query embeds", flush=True) + + # Base (no-arm) delivery baseline for delivered-recall / preservation / loss8. + base_delivered = {qid: ctx.deliver(qid) for qid in all_qids} + + results: list[dict[str, Any]] = [] + for quota in args.quotas: + knob = {"state_semantic_quota": quota} + label = f"q{quota}" + started = time.perf_counter() + + pool_recovered = 0 + by_bucket: dict[str, dict[str, int]] = {} + delivered_recovered = 0 + delivered_qids: list[str] = [] + pool_cases: dict[str, Any] = {} + for case in h5.cases: + qid = case["qid"] + delivered = ctx.deliver(qid, **knob) + admitted = _state_admitted(ctx, qid) + outcome = h5.case_pool_recovery(case, admitted) + pool_cases[qid] = outcome + slot = by_bucket.setdefault(case["bucket"], {"total": 0, "recovered": 0}) + slot["total"] += 1 + if outcome["recovered"]: + pool_recovered += 1 + slot["recovered"] += 1 + if h5.case_delivered_recall(case, delivered, base_delivered[qid]): + delivered_recovered += 1 + delivered_qids.append(qid) + + loss_unchanged = 0 + loss_changed: list[str] = [] + for _dom, qid in _LOSS_8: + if ctx.deliver(qid, **knob) == base_delivered[qid]: + loss_unchanged += 1 + else: + loss_changed.append(qid) + + disturbed: list[str] = [] + for qid in preserved_qids: + if set(base_delivered[qid]) - set(ctx.deliver(qid, **knob)): + disturbed.append(qid) + + default_latency = measure_latency(ctx, sample, {}) + knob_latency = measure_latency(ctx, sample, knob) + base_p95 = default_latency["p95_ms"] + row = { + "knob": knob, + "label": label, + "pool_recovery": pool_recovered, + "pool_recovery_by_bucket": by_bucket, + "pool_cases": pool_cases, + "delivered_recall": delivered_recovered, + "delivered_recall_qids": delivered_qids, + "loss8_unchanged": loss_unchanged, + "loss8_changed_qids": loss_changed, + "preservation_disturbed": len(disturbed), + "preservation_disturbed_qids": disturbed, + "latency": knob_latency, + "paired_default_latency": default_latency, + "latency_pct_vs_baseline": ( + (knob_latency["p95_ms"] / base_p95 - 1.0) * 100.0 if base_p95 else 0.0 + ), + "elapsed_s": round(time.perf_counter() - started, 1), + } + results.append(row) + print( + f" {label}: pool {pool_recovered}/30 | dRec {delivered_recovered}/30 " + f"loss8 {loss_unchanged}/8 pres {len(disturbed)}/{len(preserved_qids)} " + f"p95 {row['latency_pct_vs_baseline']:+.1f}% ({row['elapsed_s']}s)", + flush=True, + ) + + print("\n== SWEEP TABLE (component gate FROZEN; shipping knob NOT picked here) ==") + header = (f"{'knob':<8}{'poolRec/30':>11}{'dRec/30':>9}" + f"{'loss8/8':>9}{'pres/154':>10}{'p95 dpct':>10}") + print(header) + print("-" * len(header)) + for row in results: + print(f"{row['label']:<8}{row['pool_recovery']:>11}{row['delivered_recall']:>9}" + f"{row['loss8_unchanged']:>9}{row['preservation_disturbed']:>10}" + f"{row['latency_pct_vs_baseline']:>+9.1f}%") + + payload = { + "golden_gate": golden, + "seed_split": split, + "targets_manifest": str(args.targets), + "db_dir": str(args.db_dir), + "preserved_universe": len(preserved_qids), + "unique_query_embeds": provider.calls, + "quotas": list(args.quotas), + "sweep": results, + } + args.out.write_text(json.dumps(payload, indent=2)) + print(f"\nwrote {args.out}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarking/state_embedding_backfill.py b/benchmarking/state_embedding_backfill.py new file mode 100644 index 000000000..fd3bdcd9f --- /dev/null +++ b/benchmarking/state_embedding_backfill.py @@ -0,0 +1,219 @@ +#!/usr/bin/env python3 +"""Resumable per-state embedding backfill CLI (issue #142, Lane S / W3a). + +Embeds one vector per trajectory state (``states.text``) into the additive +``lcm_trajectory_state_embeddings`` table via ``TrajectoryStore. +build_state_semantic_index`` -- packed 32 items / 72K tokens per request, a +chunked path for states over Voyage's per-document cap, resumable (a re-run +embeds only the remainder), and metered. A JSONL spend ledger is appended after +every dispatched request so an interrupted run leaves an auditable trail, and the +run ABORTS if the projected total cost would exceed the cap. + +Usage (backfill a WORKING COPY -- never the frozen originals): + + VOYAGE_API_KEY=... python3 -m benchmarking.state_embedding_backfill \ + --db /path/to/copy/lcm.db \ + --provider voyage --model voyage-4 \ + --ledger /path/to/artifacts/W3A-backfill-web-ledger.jsonl \ + --cost-cap 10.0 +""" +from __future__ import annotations + +import argparse +import importlib.util +import json +import sqlite3 +import sys +import time +from pathlib import Path +from typing import Any + +_REPO_ROOT = Path(__file__).resolve().parent.parent + +# voyage-4 document-embedding price ($/1M tokens); mirrors command.py's +# _VOYAGE_USD_PER_MILLION_TOKENS (the table the #141 sizing used). +_VOYAGE_USD_PER_MILLION_TOKENS = { + "voyage-4-large": 0.12, + "voyage-4": 0.06, + "voyage-4-lite": 0.02, + "voyage-3": 0.06, + "voyage-3.5": 0.06, + "voyage-3-large": 0.18, +} + + +def _bootstrap_package(repo_root: Path) -> Any: + pkg = "hermes_lcm" + if pkg in sys.modules: + return sys.modules[pkg] + parent = str(repo_root.parent) + if parent not in sys.path: + sys.path.insert(0, parent) + spec = importlib.util.spec_from_file_location( + pkg, str(repo_root / "__init__.py"), + submodule_search_locations=[str(repo_root)], + ) + mod = importlib.util.module_from_spec(spec) + mod.__path__ = [str(repo_root)] + mod.__package__ = pkg + sys.modules[pkg] = mod + for py_file in repo_root.glob("*.py"): + if py_file.name == "__init__.py": + continue + sub_name = f"{pkg}.{py_file.stem}" + if sub_name in sys.modules: + continue + sub_spec = importlib.util.spec_from_file_location( + sub_name, str(py_file), submodule_search_locations=[] + ) + sub_mod = importlib.util.module_from_spec(sub_spec) + sub_mod.__package__ = pkg + sys.modules[sub_name] = sub_mod + setattr(mod, py_file.stem, sub_mod) + try: + sub_spec.loader.exec_module(sub_mod) + except Exception: + pass + return mod + + +def _open_store(db_path: Path, asset_root: Path): + ts = sys.modules["hermes_lcm.trajectory_store"] + conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True) + identity_json = json.loads( + conn.execute( + "SELECT identity_json FROM lcm_trajectory_corpora WHERE singleton=1" + ).fetchone()[0] + ) + conn.close() + identity = ts.CorpusIdentity( + dataset_name=identity_json["dataset_name"], + dataset_revision=identity_json["dataset_revision"], + harness_commit=identity_json["harness_commit"], + tier=identity_json["tier"], + domain=identity_json["domain"], + ingest_config_digest=identity_json.get("ingest_config_digest", ""), + ) + return ts.TrajectoryStore(db_path, identity, asset_root=asset_root) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--db", type=Path, required=True, + help="WORKING-COPY lcm.db to backfill (never a frozen original)") + parser.add_argument("--asset-root", type=Path, default=None) + parser.add_argument("--provider", default="voyage") + parser.add_argument("--model", default="voyage-4") + parser.add_argument("--ledger", type=Path, required=True) + parser.add_argument("--summary", type=Path, default=None) + parser.add_argument("--batch-items", type=int, default=32) + parser.add_argument("--timeout", type=float, default=300.0) + parser.add_argument("--cost-cap", type=float, default=10.0, + help="abort if projected total cost exceeds this (USD)") + parser.add_argument("--no-resume", action="store_true") + args = parser.parse_args() + + _bootstrap_package(_REPO_ROOT) + ts = sys.modules["hermes_lcm.trajectory_store"] + asset_root = args.asset_root or args.db.parent + store = _open_store(args.db, asset_root) + + provider = ts.create_trajectory_embedding_provider( + args.provider, args.model, timeout_seconds=args.timeout, for_backfill=True, + ) + rate = _VOYAGE_USD_PER_MILLION_TOKENS.get(args.model, 0.06) + args.ledger.parent.mkdir(parents=True, exist_ok=True) + started = time.perf_counter() + checkpoint_state = {"logged_50pct": False} + + def _projected(stats: dict[str, Any]) -> tuple[float, float]: + cost = stats["billed_tokens"] / 1e6 * rate + embedded = max(1, stats["states_embedded"]) + pending = max(1, stats["pending"]) + projected_tokens = stats["billed_tokens"] / embedded * pending + projected_cost = projected_tokens / 1e6 * rate + return cost, projected_cost + + class _AbortCostCap(RuntimeError): + pass + + def _callback(stats: dict[str, Any]) -> None: + cost, projected_cost = _projected(stats) + elapsed = time.perf_counter() - started + record = { + "ts": time.time(), + "db": str(args.db), + "model": args.model, + "states_embedded": stats["states_embedded"], + "pending": stats["pending"], + "chunked_states": stats["chunked_states"], + "provider_calls": stats["provider_calls"], + "billed_tokens": stats["billed_tokens"], + "cost_usd": round(cost, 6), + "projected_total_usd": round(projected_cost, 6), + "elapsed_s": round(elapsed, 1), + } + with args.ledger.open("a") as handle: + handle.write(json.dumps(record) + "\n") + done = stats["states_embedded"] + pending = stats["pending"] + if pending and not checkpoint_state["logged_50pct"] and done >= pending / 2: + checkpoint_state["logged_50pct"] = True + print(f"[CHECKPOINT 50%] {done}/{pending} states, spent ${cost:.4f}, " + f"projected total ${projected_cost:.4f} (cap ${args.cost_cap})", + flush=True) + if projected_cost > args.cost_cap: + raise _AbortCostCap( + f"projected total ${projected_cost:.2f} exceeds cap ${args.cost_cap:.2f} " + f"after {done} states (${cost:.4f} spent)" + ) + if stats["provider_calls"] % 25 == 0: + print(f" {done}/{pending} embedded, {stats['provider_calls']} calls, " + f"${cost:.4f} spent, ~${projected_cost:.4f} projected " + f"({elapsed:.0f}s)", flush=True) + + print(f"== state backfill: {args.db} (model={args.model}, rate=${rate}/M) ==", + flush=True) + try: + stats = store.build_state_semantic_index( + provider, + resume=not args.no_resume, + batch_max_items=args.batch_items, + progress_callback=_callback, + ) + except _AbortCostCap as exc: + print(f"ABORTED (cost cap): {exc}", flush=True) + store.close() + return 2 + + elapsed = time.perf_counter() - started + cost = stats["billed_tokens"] / 1e6 * rate + summary = { + "db": str(args.db), + "provider": args.provider, + "model": args.model, + "rate_usd_per_million": rate, + "profile_digest": stats["profile_digest"], + "dim": stats["dim"], + "total_states": stats["total_states"], + "already_embedded_at_start": stats["already_embedded"], + "pending_at_start": stats["pending"], + "states_embedded_this_run": stats["states_embedded"], + "chunked_states": stats["chunked_states"], + "provider_calls": stats["provider_calls"], + "billed_tokens": stats["billed_tokens"], + "cost_usd": round(cost, 6), + "runtime_s": round(elapsed, 1), + "status": stats["status"], + } + store.close() + if args.summary is not None: + args.summary.parent.mkdir(parents=True, exist_ok=True) + args.summary.write_text(json.dumps(summary, indent=2)) + print("== DONE ==", flush=True) + print(json.dumps(summary, indent=2), flush=True) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From c9fc5580eed386cd6a7bc4fcdb2f50c572f0a225 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 10:43:57 +0700 Subject: [PATCH 17/54] bench: fix state-semantic sweep harness bootstrap + rate guard (#142) Bootstrap the hermes_lcm package before constructing the query provider (the provider import needs it), and disable the interactive per-minute call-rate guard on the sweep's query embedder (the 451-query warm pass is the bulk pattern for_backfill exempts, and tripped ProviderRateLimited otherwise). Sweep runs clean end-to-end: golden 451/451, quotas 4-64. --- benchmarking/h5_state_semantic_replay.py | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/benchmarking/h5_state_semantic_replay.py b/benchmarking/h5_state_semantic_replay.py index c12f7ca4f..85b99b38d 100644 --- a/benchmarking/h5_state_semantic_replay.py +++ b/benchmarking/h5_state_semantic_replay.py @@ -40,7 +40,8 @@ sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from benchmarking.h3_composition_replay import ( # noqa: E402 - _H1, _H31, _RUN_ROOT, ReplayContext, golden_gate, measure_latency, + _H1, _H31, _REPO_ROOT, _RUN_ROOT, ReplayContext, _bootstrap_package, + golden_gate, measure_latency, ) from benchmarking.h5_recall_replay import ( # noqa: E402 _LOSS_8, H5Context, @@ -59,9 +60,14 @@ class CachedVoyageQueryProvider: model_id = "voyage-4" def __init__(self, model: str = "voyage-4", timeout: float = 30.0) -> None: - from hermes_lcm.embedding_provider import VoyageProvider + from hermes_lcm.embedding_provider import EmbeddingSpendGuard, VoyageProvider - self._real = VoyageProvider(model, timeout=timeout) + # Disable the interactive per-minute call-rate guard: this offline sweep + # embeds the 451 unique queries in a tight loop (each is cheap, ~cents + # total), which is exactly the bulk pattern for_backfill=True exempts. + self._real = VoyageProvider( + model, timeout=timeout, spend_guard=EmbeddingSpendGuard(max_calls=0) + ) self.model_id = model self._cache: dict[str, list[float]] = {} self.last_usage_tokens = 0 @@ -146,6 +152,7 @@ def main() -> int: flush=True) return 3 + _bootstrap_package(_REPO_ROOT) provider = CachedVoyageQueryProvider() ctx = StateReplayContext( args.run_root, args.h1_artifacts, args.h31_artifacts, args.db_dir, provider From 505a4d88611f86c91e663986ef5d7e90a5a5aed6 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 11:20:26 +0700 Subject: [PATCH 18/54] feat: add compact retrieval delivery controls --- benchmarking/w3b_diversity_replay.py | 456 ++++++++++++++++ tests/test_trajectory_compact_delivery.py | 251 +++++++++ trajectory_store.py | 630 +++++++++++++++++++++- 3 files changed, 1325 insertions(+), 12 deletions(-) create mode 100644 benchmarking/w3b_diversity_replay.py create mode 100644 tests/test_trajectory_compact_delivery.py diff --git a/benchmarking/w3b_diversity_replay.py b/benchmarking/w3b_diversity_replay.py new file mode 100644 index 000000000..e8353bfb6 --- /dev/null +++ b/benchmarking/w3b_diversity_replay.py @@ -0,0 +1,456 @@ +#!/usr/bin/env python3 +"""Provider-free W3b C1 and W3a-state composition replay. + +This driver reuses the frozen H5 target/preservation contract. The standalone +C1 cells run on the original read-only replay DBs. The q16/q32 composition cells +run on the read-only W3a backfilled working copies. + +No embedding or reader provider is called. For the state-semantic cells, the +query direction is reconstructed deterministically from: + +* the recorded source-candidate cosine scores in the frozen H1 trace; and +* the corresponding stored source vectors in the W3a DB. + +The minimum-norm vector satisfying those recorded projections is used only to +rank the already-backfilled state vectors. This is provider-free component +evidence, not a replacement for an official reader/judge run. +""" +from __future__ import annotations + +import argparse +import json +import sqlite3 +import struct +import sys +import time +from pathlib import Path +from typing import Any + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from benchmarking.h3_composition_replay import ( # noqa: E402 + _H1, + _H31, + _RUN_ROOT, + ReplayContext, + golden_gate, +) +from benchmarking.h5_recall_replay import ( # noqa: E402 + _H5_TARGETS, + _LOSS_8, + H5Context, +) + +_DEFAULT_DB_DIR = Path("/Volumes/LEXAR/hermes-work/W3A-dbwork") + + +class ProjectedQueryProvider: + """Reconstruct query vectors from frozen source-vector projections.""" + + provider_id = "voyage" + model_id = "voyage-4" + + def __init__( + self, + context: "ProjectedStateReplayContext", + domain: str, + db_path: Path, + ) -> None: + self._context = context + self._domain = domain + self._cache: dict[str, list[float]] = {} + self.last_usage_tokens = 0 + self.external_calls = 0 + self.projections_built = 0 + connection = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True) + connection.row_factory = sqlite3.Row + profile = connection.execute( + """ + SELECT profile_digest, dim + FROM lcm_trajectory_embedding_profiles + WHERE active = 1 + """ + ).fetchone() + if profile is None: + connection.close() + raise RuntimeError(f"{domain}: active source embedding profile missing") + self._dim = int(profile["dim"]) + rows = connection.execute( + """ + SELECT source_id, vector + FROM lcm_trajectory_embeddings + WHERE profile_digest = ? + """, + (str(profile["profile_digest"]),), + ).fetchall() + connection.close() + self._source_vectors = { + int(row["source_id"]): struct.unpack( + f"<{self._dim}f", bytes(row["vector"]) + ) + for row in rows + } + if not self._source_vectors: + raise RuntimeError(f"{domain}: source vectors missing") + + @staticmethod + def _solve(gram: list[list[float]], target: list[float]) -> list[float]: + """Gaussian elimination with deterministic pivoting and tiny ridge.""" + size = len(target) + augmented = [ + [ + gram[row][column] + (1e-9 if row == column else 0.0) + for column in range(size) + ] + [target[row]] + for row in range(size) + ] + for column in range(size): + pivot = max( + range(column, size), + key=lambda row: abs(augmented[row][column]), + ) + if abs(augmented[pivot][column]) < 1e-12: + raise RuntimeError("recorded source projection matrix is singular") + augmented[column], augmented[pivot] = ( + augmented[pivot], + augmented[column], + ) + divisor = augmented[column][column] + augmented[column] = [ + value / divisor for value in augmented[column] + ] + for row in range(size): + if row == column: + continue + factor = augmented[row][column] + if factor == 0.0: + continue + augmented[row] = [ + left - factor * right + for left, right in zip( + augmented[row], augmented[column] + ) + ] + return [augmented[row][-1] for row in range(size)] + + def embed_query(self, text: str) -> list[float]: + cached = self._cache.get(text) + if cached is not None: + return cached + qids = [ + qid + for qid, (domain, question) in self._context.questions.items() + if domain == self._domain and question == text + ] + if len(qids) != 1: + raise RuntimeError( + f"{self._domain}: expected one frozen qid for query, got {qids}" + ) + ranks = self._context._injected_ranks(qids[0]) + matrix_rows: list[tuple[float, ...]] = [] + scores: list[float] = [] + for source_id, score in ranks: + vector = self._source_vectors.get(source_id) + if vector is None: + continue + matrix_rows.append(vector) + scores.append(float(score)) + if not matrix_rows: + raise RuntimeError(f"{self._domain}/{qids[0]}: no source projections") + gram = [ + [ + sum(left * right for left, right in zip(row_a, row_b)) + for row_b in matrix_rows + ] + for row_a in matrix_rows + ] + coefficients = self._solve(gram, scores) + vector = [ + sum( + coefficients[row] * matrix_rows[row][column] + for row in range(len(matrix_rows)) + ) + for column in range(self._dim) + ] + norm = sum(value * value for value in vector) ** 0.5 + if norm <= 0.0 or len(vector) != self._dim: + raise RuntimeError( + f"{self._domain}/{qids[0]}: invalid reconstructed query vector" + ) + cached = [float(value) for value in vector] + self._cache[text] = cached + self.projections_built += 1 + return cached + + def embed_documents(self, texts): # noqa: ARG002 + raise RuntimeError("ProjectedQueryProvider is query-only") + + +class ProjectedStateReplayContext(ReplayContext): + """W3a DB replay with provider-free reconstructed state-query vectors.""" + + def __init__(self, run_root, h1, h31, db_dir: Path) -> None: + self._db_dir = Path(db_dir) + self.providers: dict[str, ProjectedQueryProvider] = {} + super().__init__(run_root, h1, h31) + + def _open_store(self, db_path, domain): # noqa: ARG002 + real_db = self._db_dir / f"{domain}.lcm.db" + connection = sqlite3.connect(f"file:{real_db}?mode=ro", uri=True) + identity_json = json.loads( + connection.execute( + "SELECT identity_json FROM lcm_trajectory_corpora WHERE singleton=1" + ).fetchone()[0] + ) + connection.close() + identity = self._ts.CorpusIdentity( + dataset_name=identity_json["dataset_name"], + dataset_revision=identity_json["dataset_revision"], + harness_commit=identity_json["harness_commit"], + tier=identity_json["tier"], + domain=identity_json["domain"], + ingest_config_digest=identity_json.get("ingest_config_digest", ""), + ) + base = self._ts.TrajectoryStore + + class _ReplayStore(base): # type: ignore[misc, valid-type] + injected: list[tuple[int, float]] = [] + + def _semantic_source_ranks(self, query: str): # noqa: ARG002 + return list(self.injected) + + provider = ProjectedQueryProvider(self, domain, real_db) + self.providers[domain] = provider + return _ReplayStore( + real_db, + identity, + asset_root=real_db.parent, + read_only=True, + semantic_top_trajectories=12, + embedding_provider=provider, + ) + + +def _pool_ids(context: ReplayContext, qid: str) -> set[int]: + domain, _question = context.questions[qid] + telemetry = context.stores[domain].last_query_telemetry() or {} + cap = telemetry.get("diversity_cap") or {} + return {int(value) for value in cap.get("survivor_state_ids", [])} + + +def _target_in_pool(h5: H5Context, case: dict[str, Any], pool: set[int]) -> bool: + domain = str(case["domain"]) + if case["status"] == "pinned": + target_ids = { + int(value) + for target in case["targets"] + for value in target["state_ids"].values() + } + return bool(target_ids & pool) + if case["status"] == "trajectory": + return any( + set(h5.traj_states(domain, target["trajectory_id"]).values()) & pool + for target in case["targets"] + ) + return False + + +def _evaluate_cell( + context: ReplayContext, + h5: H5Context, + kwargs: dict[str, Any], + baseline_delivered: dict[str, list[str]], + preserved_qids: list[str], +) -> dict[str, Any]: + pool_covered = 0 + pool_new = 0 + delivered_recall = 0 + delivered_qids: list[str] = [] + default_pool_covered = 0 + capped_out = 0 + for case in h5.cases: + qid = str(case["qid"]) + delivered = context.deliver(qid, **kwargs) + pool = _pool_ids(context, qid) + default_pool = h5.default_pool_ids(qid) + covered = _target_in_pool(h5, case, pool) + covered_default = _target_in_pool(h5, case, default_pool) + pool_covered += int(covered) + default_pool_covered += int(covered_default) + pool_new += int(covered and not covered_default) + if h5.case_delivered_recall( + case, delivered, baseline_delivered[qid] + ): + delivered_recall += 1 + delivered_qids.append(qid) + domain, _question = context.questions[qid] + telemetry = context.stores[domain].last_query_telemetry() or {} + capped_out += int( + (telemetry.get("diversity_cap") or {}).get("capped_out", 0) + ) + + disturbed: list[str] = [] + for qid in preserved_qids: + before = set(baseline_delivered[qid]) + after = set(context.deliver(qid, **kwargs)) + if before - after: + disturbed.append(qid) + loss_changed: list[str] = [] + for _domain, qid in _LOSS_8: + if context.deliver(qid, **kwargs) != baseline_delivered[qid]: + loss_changed.append(qid) + return { + "kwargs": kwargs, + "target_pool_covered": pool_covered, + "default_target_pool_covered": default_pool_covered, + "new_pool_recovery": pool_new, + "delivered_recall_at_16": delivered_recall, + "delivered_recall_qids": delivered_qids, + "preservation_retained": len(preserved_qids) - len(disturbed), + "preservation_disturbed": len(disturbed), + "preservation_disturbed_qids": disturbed, + "loss8_unchanged": len(_LOSS_8) - len(loss_changed), + "loss8_changed_qids": loss_changed, + "capped_out_target_queries_total": capped_out, + } + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--run-root", type=Path, default=_RUN_ROOT) + parser.add_argument("--h1-artifacts", type=Path, default=_H1) + parser.add_argument("--h31-artifacts", type=Path, default=_H31) + parser.add_argument("--targets", type=Path, default=_H5_TARGETS) + parser.add_argument("--db-dir", type=Path, default=_DEFAULT_DB_DIR) + parser.add_argument("--out", type=Path, required=True) + parser.add_argument("--caps", type=int, nargs="+", default=[1, 2, 3]) + parser.add_argument("--state-quotas", type=int, nargs="+", default=[16, 32]) + args = parser.parse_args() + + args.out.parent.mkdir(parents=True, exist_ok=True) + base_context = ReplayContext( + args.run_root, args.h1_artifacts, args.h31_artifacts + ) + state_context = ProjectedStateReplayContext( + args.run_root, args.h1_artifacts, args.h31_artifacts, args.db_dir + ) + base_h5 = H5Context(base_context, args.targets) + state_h5 = H5Context(state_context, args.targets) + + print("== GOLDEN GATES (all W3b knobs off) ==", flush=True) + base_golden = golden_gate(base_context) + state_golden = golden_gate(state_context) + print( + f" base {base_golden['passed']}/{base_golden['total']} | " + f"state-db {state_golden['passed']}/{state_golden['total']}", + flush=True, + ) + if ( + base_golden["passed"] != base_golden["total"] + or state_golden["passed"] != state_golden["total"] + ): + print("GOLDEN GATE FAILED -- aborting before cells", flush=True) + return 1 + + reconciliation = json.loads( + ( + args.h1_artifacts + / "old-rescore" + / "reconciliation_sets.json" + ).read_text() + ) + preserved_qids = sorted({ + item.split("/", 1)[1] for item in reconciliation["preserved"] + }) + all_qids = sorted({ + *preserved_qids, + *(str(case["qid"]) for case in base_h5.cases), + *(qid for _domain, qid in _LOSS_8), + }) + base_delivered = { + qid: base_context.deliver(qid) for qid in all_qids + } + state_base_delivered = { + qid: state_context.deliver(qid) for qid in all_qids + } + if base_delivered != state_base_delivered: + print("STATE-DB BASELINE DIFFERS -- aborting composition cells", flush=True) + return 2 + + cells: list[dict[str, Any]] = [] + for cap in args.caps: + started = time.perf_counter() + row = _evaluate_cell( + base_context, + base_h5, + {"diversity_cap": cap}, + base_delivered, + preserved_qids, + ) + row["label"] = f"cap{cap}" + row["state_query_mode"] = "not_used" + row["elapsed_s"] = round(time.perf_counter() - started, 3) + cells.append(row) + print( + f" {row['label']}: pool {row['target_pool_covered']}/30 " + f"(new {row['new_pool_recovery']}) | " + f"dRec@16 {row['delivered_recall_at_16']}/30 | " + f"pres {row['preservation_retained']}/154", + flush=True, + ) + for quota in args.state_quotas: + for cap in args.caps: + started = time.perf_counter() + row = _evaluate_cell( + state_context, + state_h5, + { + "state_semantic_quota": quota, + "diversity_cap": cap, + }, + base_delivered, + preserved_qids, + ) + row["label"] = f"q{quota}xcap{cap}" + row["state_query_mode"] = "recorded-source-projection" + row["elapsed_s"] = round(time.perf_counter() - started, 3) + cells.append(row) + print( + f" {row['label']}: pool {row['target_pool_covered']}/30 " + f"(new {row['new_pool_recovery']}) | " + f"dRec@16 {row['delivered_recall_at_16']}/30 | " + f"pres {row['preservation_retained']}/154", + flush=True, + ) + + payload = { + "proof_boundary": ( + "Provider-free component replay only; reconstructed query directions " + "do not prove official reader/judge accuracy or promotion." + ), + "golden_gate": { + "base": base_golden, + "state_db": state_golden, + }, + "targets_manifest": str(args.targets), + "db_dir": str(args.db_dir), + "caps": list(args.caps), + "state_quotas": list(args.state_quotas), + "preserved_universe": len(preserved_qids), + "external_provider_calls": sum( + provider.external_calls + for provider in state_context.providers.values() + ), + "query_projections_built": { + domain: provider.projections_built + for domain, provider in state_context.providers.items() + }, + "cells": cells, + } + args.out.write_text(json.dumps(payload, indent=2) + "\n") + print(f"wrote {args.out}", flush=True) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_trajectory_compact_delivery.py b/tests/test_trajectory_compact_delivery.py new file mode 100644 index 000000000..2dd1dc3f6 --- /dev/null +++ b/tests/test_trajectory_compact_delivery.py @@ -0,0 +1,251 @@ +"""W3b compact-delivery components: C1 diversity, C2 excerpts, C3 compilation.""" + +from __future__ import annotations + +import hashlib +from pathlib import Path + +from hermes_lcm.tokens import count_tokens +from hermes_lcm.trajectory_store import ( + CorpusIdentity, + TrajectorySource, + TrajectoryState, + TrajectoryStore, +) + + +class StateProvider: + provider_id = "fake" + model_id = "fake-state-v1" + dim = 3 + + def __init__(self) -> None: + self.last_usage_tokens = 0 + + @staticmethod + def _vector(text: str) -> list[float]: + folded = str(text).casefold() + if "alpha-answer" in folded: + return [1.0, 0.0, 0.0] + if "beta-answer" in folded: + return [0.0, 1.0, 0.0] + return [0.0, 0.0, 1.0] + + def embed_documents(self, texts): + self.last_usage_tokens = len(texts) + return [self._vector(text) for text in texts] + + def embed_query(self, text): # noqa: ARG002 + self.last_usage_tokens = 1 + return [1.0, 0.0, 0.0] + + +def _identity(domain: str = "web") -> CorpusIdentity: + return CorpusIdentity( + dataset_name="example/w3b", + dataset_revision="rev-w3b", + harness_commit="harness-w3b-1", + tier="small", + domain=domain, + ingest_config_digest="w3b-test-v1", + ) + + +def _source( + asset_root: Path, + *, + trajectory_id: str, + ordinal: int, + goal: str, + texts: tuple[str, ...], + urls: tuple[str, ...] | None = None, +) -> TrajectorySource: + states = [] + for index, text in enumerate(texts): + screenshot = asset_root / f"{trajectory_id}-{index}.png" + screenshot.write_bytes(b"png" + hashlib.sha256(text.encode()).digest()) + states.append(TrajectoryState( + state_index=index, + step=index, + url=( + urls[index] + if urls is not None + else f"https://example.test/{trajectory_id}/{index}" + ), + incoming_action=None if index == 0 else f"click step {index}", + thoughts=f"inspect state {index}", + text=text, + screenshot_path=screenshot, + )) + return TrajectorySource( + trajectory_id=trajectory_id, + ordinal=ordinal, + goal=goal, + start_url=f"https://example.test/{trajectory_id}", + outcome="completed", + states=tuple(states), + source_payload={"id": trajectory_id, "goal": goal}, + ) + + +def _build_store(tmp_path: Path, *, provider=None) -> TrajectoryStore: + asset_root = tmp_path / "assets" + asset_root.mkdir() + store = TrajectoryStore( + tmp_path / "lcm.db", + _identity(), + asset_root=asset_root, + embedding_provider=provider, + ) + store.insert(_source( + asset_root, + trajectory_id="hub", + ordinal=0, + goal="Generic export dashboard", + texts=tuple( + f"widget configuration export repeated dashboard panel {index}" + for index in range(7) + ), + )) + store.insert(_source( + asset_root, + trajectory_id="target", + ordinal=1, + goal="Problem List Cleanup", + texts=( + "widget export procedure overview", + "alpha-answer duplicate description field exact instruction", + ), + )) + store.insert(_source( + asset_root, + trajectory_id="other", + ordinal=2, + goal="Open another report", + texts=("widget export report toolbar",), + )) + store.finalize(["hub", "target", "other"]) + return store + + +def test_all_w3b_knobs_default_off_preserve_bytes_and_telemetry(tmp_path: Path): + store = _build_store(tmp_path) + query = "widget configuration export" + baseline = store.query(query, image_limit=0) + baseline_payload = [hit.to_dict() for hit in baseline] + baseline_telemetry = store.last_query_telemetry() + + explicit_off = store.query( + query, + image_limit=0, + diversity_cap=0, + adaptive_excerpt=False, + sharp_token_budget=0, + ) + assert [hit.to_dict() for hit in explicit_off] == baseline_payload + assert store.last_query_telemetry() == baseline_telemetry + assert "diversity_cap" not in baseline_telemetry + assert "adaptive_excerpt" not in baseline_telemetry + assert "sharp_compilation" not in baseline_telemetry + + +def test_c1_caps_after_state_arm_composition_and_reports_trajectory(tmp_path: Path): + provider = StateProvider() + store = _build_store(tmp_path, provider=provider) + store.build_state_semantic_index(provider) + + hits = store.query( + "widget configuration export", + image_limit=0, + include_adjacent=False, + state_semantic_quota=8, + diversity_cap=2, + ) + per_trajectory: dict[str, int] = {} + for hit in hits: + per_trajectory[hit.trajectory_id] = ( + per_trajectory.get(hit.trajectory_id, 0) + 1 + ) + assert max(per_trajectory.values()) <= 2 + assert any("alpha-answer" in hit.text for hit in hits) + + telemetry = store.last_query_telemetry() + assert telemetry["state_semantic_expansion"]["admitted"] + cap = telemetry["diversity_cap"] + assert cap["cap"] == 2 + hub = next( + entry for entry in cap["trajectories"] + if entry["trajectory_id"] == "hub" + ) + assert hub["before"] == 7 + assert hub["after"] == 2 + assert hub["capped_out"] == 5 + + +def test_c2_densest_window_anchors_rare_query_term_and_shifts_budget(): + prefix = ("Low Stock Report dashboard navigation report " * 100) + needle = "TABLE HEADER Quantity Source Code Scope" + suffix = (" report footer " * 100) + text = prefix + needle + suffix + excerpt, offset = TrajectoryStore._densest_exact_excerpt( + text, + "Which column is next to Quantity in the Low Stock Report?", + 500, + ) + assert "Quantity Source Code" in excerpt + assert offset > 0 + + rows = [ + { + "state_id": 1, + "trajectory_id": "repeat", + "text": "x" * 4_000, + }, + { + "state_id": 2, + "trajectory_id": "repeat", + "text": "y" * 4_000, + }, + { + "state_id": 3, + "trajectory_id": "sole", + "text": "z" * 4_000, + }, + ] + limits, telemetry = TrajectoryStore._adaptive_excerpt_limits( + rows, 1_000, True + ) + assert limits[1] == limits[2] == 750 + assert limits[3] == 1_500 + assert sum(limits.values()) == 3_000 + assert telemetry["raised_only_hits"] == [{"state_id": 3, "chars": 1500}] + + +def test_c3_typed_queries_exact_title_template_and_budget(tmp_path: Path): + store = _build_store(tmp_path) + budget = 900 + hits = store.query( + ( + "According to company protocol `Problem List Cleanup`, what should " + "we change in the duplicate description field before deletion?" + ), + image_limit=0, + include_adjacent=False, + sharp_token_budget=budget, + ) + assert hits + assert hits[0].trajectory_id == "target" + rendered_tokens = sum( + count_tokens(TrajectoryStore._rendered_hit_text(hit)) for hit in hits + ) + assert rendered_tokens <= budget + + sharp = store.last_query_telemetry()["sharp_compilation"] + assert sharp["question_template"] == "procedure" + assert {entry["pool_type"] for entry in sharp["subqueries"]} >= { + "raw_state", + "entity", + "action", + } + assert sharp["exact_title_boosts"] + assert sharp["rendered_text_tokens_after"] <= budget diff --git a/trajectory_store.py b/trajectory_store.py index f233c93ad..ba790e61c 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -9,7 +9,7 @@ from __future__ import annotations from collections import deque -from dataclasses import asdict, dataclass +from dataclasses import asdict, dataclass, replace import hashlib import json import math @@ -53,6 +53,8 @@ _MAX_IMAGES = 8 _MAX_QUERY_TEXT_CHARS = 8_000 _MAX_ADJACENCY_RADIUS = 8 +_MAX_DIVERSITY_CAP = 24 +_MAX_SHARP_TOKEN_BUDGET = 4_000 _MAX_SOURCE_JSON_CHARS = 16_000_000 _MAX_TEXT_CHARS = 2_000_000 _MAX_SEMANTIC_DOCUMENT_CHARS = 48_000 @@ -64,6 +66,17 @@ "it", "of", "on", "or", "should", "the", "then", "to", "was", "were", "what", "when", "where", "which", "who", "why", "with", "would", }) +_TEMPORAL_TERMS = frozenset({ + "after", "before", "current", "date", "day", "earliest", "first", + "initial", "last", "latest", "month", "newest", "previous", "recent", + "time", "timestamp", "today", "week", "when", "year", "yesterday", +}) +_ACTION_TERMS = frozenset({ + "add", "apply", "assign", "buy", "change", "choose", "click", "compare", + "configure", "create", "delete", "edit", "export", "filter", "insert", + "navigate", "open", "order", "remove", "retry", "save", "search", + "select", "set", "sort", "submit", "update", "view", +}) _EXACT_REF_RE = re.compile( r"^trajectory://(?P[0-9a-f]{64})/" r"(?P[^/]+)/state/(?P[0-9]+)$" @@ -1953,6 +1966,7 @@ def _row_to_hit( include_image: bool, query: str | None = None, text_char_limit: int | None = None, + dense_excerpt: bool = False, ) -> TrajectoryHit: corpus_uid = self.corpus_uid if not corpus_uid: @@ -1971,7 +1985,10 @@ def _row_to_hit( text_offset = 0 text_truncated = False else: - text, text_offset = self._exact_excerpt( + excerpt_fn = ( + self._densest_exact_excerpt if dense_excerpt else self._exact_excerpt + ) + text, text_offset = excerpt_fn( full_text, query or "", text_char_limit, @@ -2033,12 +2050,494 @@ def _exact_excerpt(text: str, query: str, limit: int) -> tuple[str, int]: return text[start:end], start @staticmethod - def _select_diverse(rows: Iterable[sqlite3.Row], limit: int) -> list[sqlite3.Row]: + def _densest_exact_excerpt(text: str, query: str, limit: int) -> tuple[str, int]: + """Return the deterministic window with the densest rare query terms. + + The historical excerpt anchors on the first query match. Web AXTree + pages often repeat generic page-title terms near the top while the + answer-bearing field occurs below the fold. This scorer gives rarer + query terms more weight, then uses unique-term coverage, occurrence + count, and earliest offset as deterministic tie-breaks. + """ + if len(text) <= limit: + return text, 0 + folded = text.casefold() + terms: list[str] = [] + seen: set[str] = set() + for raw in extract_search_terms(query): + term = raw.casefold().strip() + if len(term) < 2 or term in _STOPWORDS or term in seen: + continue + seen.add(term) + terms.append(term) + occurrences: list[tuple[int, str, float]] = [] + for term in terms: + positions = [ + match.start() for match in re.finditer(re.escape(term), folded) + ] + if not positions: + continue + weight = 1.0 / len(positions) + occurrences.extend((position, term, weight) for position in positions) + if not occurrences: + return text[:limit], 0 + occurrences.sort(key=lambda item: (item[0], item[1])) + best: tuple[float, int, int, int] | None = None + right = 0 + counts: dict[str, int] = {} + weighted = 0.0 + for left, (left_pos, _left_term, _left_weight) in enumerate(occurrences): + while right < len(occurrences) and occurrences[right][0] - left_pos < limit: + _position, term, weight = occurrences[right] + if counts.get(term, 0) == 0: + weighted += weight + counts[term] = counts.get(term, 0) + 1 + right += 1 + score = (weighted, len(counts), right - left, -left_pos) + if best is None or score > best: + best = score + # Anchor on the rarest term inside the winning window, leaving + # two thirds of the excerpt after it for answer values/columns. + anchor = max( + occurrences[left:right], + key=lambda item: (item[2], item[0]), + )[0] + best_start = max(0, anchor - (limit // 3)) + _position, term, weight = occurrences[left] + counts[term] -= 1 + if counts[term] == 0: + del counts[term] + weighted -= weight + start = min(best_start, len(text) - limit) + return text[start:start + limit], start + + @staticmethod + def _candidate_term_set(row: Any) -> set[str]: + """Bounded lexical signature used by the per-trajectory MMR pass.""" + text = " ".join( + str(row[key] or "") + for key in ("goal", "url", "incoming_action", "text") + )[:12_000] + return { + token + for token in re.findall(r"[a-z0-9][a-z0-9_-]+", text.casefold()) + if len(token) >= 3 and token not in _STOPWORDS + } + + @staticmethod + def _cap_composed_pool( + rows: Sequence[Any], + cap: int, + ) -> tuple[list[Any], dict[str, Any]]: + """Apply one global per-trajectory cap after all candidate arms compose. + + Survivors are chosen independently within each trajectory by a bounded + MMR-style relevance/redundancy score. The returned pool preserves the + original cross-trajectory order, so the cap changes only which states + survive, not unrelated tie-breaking. + """ + unique_rows: list[Any] = [] + seen_state_ids: set[int] = set() + for row in rows: + state_id = int(row["state_id"]) + if state_id in seen_state_ids: + continue + seen_state_ids.add(state_id) + unique_rows.append(row) + grouped: dict[str, list[tuple[int, Any]]] = {} + for position, row in enumerate(unique_rows): + grouped.setdefault(str(row["trajectory_id"]), []).append((position, row)) + survivor_ids: set[int] = set() + details: list[dict[str, Any]] = [] + for trajectory_id, candidates in grouped.items(): + if len(candidates) <= cap: + survivor_ids.update(int(row["state_id"]) for _position, row in candidates) + continue + signatures = { + int(row["state_id"]): TrajectoryStore._candidate_term_set(row) + for _position, row in candidates + } + selected: list[tuple[int, Any]] = [] + remaining = list(candidates) + while remaining and len(selected) < cap: + best_item: tuple[int, Any] | None = None + best_key: tuple[float, int, int] | None = None + for local_position, item in enumerate(remaining): + original_position, row = item + relevance = 1.0 - ( + original_position / max(1, len(unique_rows) - 1) + ) + signature = signatures[int(row["state_id"])] + redundancy = 0.0 + for _selected_position, selected_row in selected: + selected_signature = signatures[int(selected_row["state_id"])] + union = signature | selected_signature + similarity = ( + len(signature & selected_signature) / len(union) + if union else 0.0 + ) + redundancy = max(redundancy, similarity) + mmr_score = (0.72 * relevance) - (0.28 * redundancy) + key = (mmr_score, -original_position, -int(row["state_id"])) + if best_key is None or key > best_key: + best_key = key + best_item = item + assert best_item is not None + selected.append(best_item) + remaining.remove(best_item) + selected_ids = [int(row["state_id"]) for _position, row in selected] + removed_ids = [ + int(row["state_id"]) + for _position, row in candidates + if int(row["state_id"]) not in set(selected_ids) + ] + survivor_ids.update(selected_ids) + details.append({ + "trajectory_id": trajectory_id, + "before": len(candidates), + "after": len(selected_ids), + "capped_out": len(removed_ids), + "selected_state_ids": selected_ids, + "removed_state_ids": removed_ids, + }) + filtered = [ + row for row in unique_rows if int(row["state_id"]) in survivor_ids + ] + return filtered, { + "cap": cap, + "pool_before": len(unique_rows), + "pool_after": len(filtered), + "capped_out": len(unique_rows) - len(filtered), + "trajectories": details, + "survivor_state_ids": [int(row["state_id"]) for row in filtered], + } + + @staticmethod + def _exact_query_phrases(query: str) -> list[str]: + phrases: list[str] = [] + seen: set[str] = set() + for match in re.finditer(r"`([^`]+)`|\"([^\"]+)\"|'([^']+)'", query): + phrase = next(group for group in match.groups() if group is not None) + normalized = " ".join(phrase.casefold().split()) + if len(normalized) < 3 or normalized in seen: + continue + seen.add(normalized) + phrases.append(normalized) + return phrases[:8] + + @staticmethod + def _typed_subqueries(query: str) -> list[dict[str, str]]: + """Deterministically decompose one question into typed FTS queries.""" + words = re.findall(r"[A-Za-z0-9][A-Za-z0-9_-]*", query) + folded_words = [word.casefold() for word in words] + candidates: list[tuple[str, str]] = [("raw_state", query)] + exact = TrajectoryStore._exact_query_phrases(query) + if exact: + candidates.append(("entity", " ".join(exact))) + entity_clauses = [ + clause.strip() + for clause in re.split(r"\s+(?:vs\.?|versus)\s+|[,;]", query) + if 2 <= len(clause.split()) <= 18 + ] + if len(entity_clauses) >= 2: + candidates.extend(("entity", clause) for clause in entity_clauses[:3]) + temporal: list[str] = [] + actions: list[str] = [] + for index, word in enumerate(folded_words): + if word in _TEMPORAL_TERMS: + temporal.extend(folded_words[max(0, index - 1):index + 2]) + if word in _ACTION_TERMS: + actions.extend(folded_words[index:index + 3]) + if temporal: + candidates.append(("time", " ".join(temporal))) + if actions: + candidates.append(("action", " ".join(actions))) + result: list[dict[str, str]] = [] + seen: set[str] = set() + for pool_type, subquery in candidates: + normalized = " ".join(subquery.split()) + key = normalized.casefold() + if not normalized or key in seen: + continue + seen.add(key) + result.append({"pool_type": pool_type, "query": normalized}) + if len(result) >= 6: + break + return result + + def _sharp_fts_rows( + self, + query: str, + candidate_limit: int, + ) -> tuple[list[sqlite3.Row], dict[str, Any]]: + subqueries = self._typed_subqueries(query) + row_by_id: dict[int, sqlite3.Row] = {} + score_by_id: dict[int, float] = {} + exact_phrases = self._exact_query_phrases(query) + weights = {"raw_state": 1.0, "entity": 1.2, "time": 1.1, "action": 1.1} + for entry in subqueries: + expression = self._fts_expression(entry["query"]) + if not expression: + continue + rows = self._fts_rows(expression, candidate_limit) + weight = weights.get(entry["pool_type"], 1.0) + for rank, row in enumerate(rows, start=1): + state_id = int(row["state_id"]) + row_by_id[state_id] = row + score_by_id[state_id] = score_by_id.get(state_id, 0.0) + ( + weight / (60.0 + rank) + ) + boosted: list[dict[str, Any]] = [] + for state_id, row in row_by_id.items(): + title_text = f"{row['goal']} {row['url']}".casefold() + matches = [phrase for phrase in exact_phrases if phrase in title_text] + if matches: + score_by_id[state_id] += 0.05 * len(matches) + boosted.append({"state_id": state_id, "phrases": matches}) + ordered = sorted( + row_by_id.values(), + key=lambda row: ( + -score_by_id[int(row["state_id"])], + int(row["ordinal"]), + int(row["sequence_ordinal"]), + ), + )[:candidate_limit] + return ordered, { + "subqueries": subqueries, + "exact_title_boosts": boosted, + "lexical_pool_size": len(ordered), + } + + @staticmethod + def _question_template(query: str) -> str: + folded = query.casefold() + if any(term in folded for term in ("protocol", "workflow", "procedure", "steps")): + return "procedure" + if any(term in folded for term in _TEMPORAL_TERMS): + return "temporal" + if ( + any(character in query for character in ",;") + or " among " in folded + or " across " in folded + or " both " in folded + ): + return "multi_entity" + if any( + term in folded + for term in ("page", "url", "tab", "button", "column", "field", "link") + ): + return "navigation" + return "generic" + + @staticmethod + def _template_order( + rows: Sequence[sqlite3.Row], + query: str, + template: str, + ) -> list[sqlite3.Row]: + """Apply small question-type priors without replacing retrieval rank.""" + query_terms = { + term.casefold() + for term in extract_search_terms(query) + if len(term.strip()) >= 2 and term.casefold() not in _STOPWORDS + } + exact_phrases = TrajectoryStore._exact_query_phrases(query) + + def _key(item: tuple[int, sqlite3.Row]) -> tuple[float, int, int]: + position, row = item + goal_url = f"{row['goal']} {row['url']}".casefold() + action = str(row["incoming_action"] or "").casefold() + text = str(row["text"] or "")[:4_000].casefold() + exact = sum(1 for phrase in exact_phrases if phrase in goal_url) + goal_density = sum(1 for term in query_terms if term in goal_url) + action_density = sum(1 for term in query_terms if term in action) + text_density = sum(1 for term in query_terms if term in text) + prior = exact * 8.0 + if template in {"navigation", "procedure"}: + prior += goal_density * 0.6 + action_density * 0.4 + elif template == "temporal": + prior += ( + (1.0 if row["observed_at"] is not None else 0.0) + + (1.0 if row["occurred_at"] is not None else 0.0) + + text_density * 0.15 + ) + elif template == "multi_entity": + prior += goal_density * 0.35 + return (-prior, position, int(row["state_id"])) + + return [ + row for _position, row in sorted(enumerate(rows), key=_key) + ] + + @staticmethod + def _adaptive_excerpt_limits( + rows: Sequence[sqlite3.Row], + base_limit: int, + enabled: bool, + ) -> tuple[dict[int, int], dict[str, Any] | None]: + limits = {int(row["state_id"]): base_limit for row in rows} + if not enabled or not rows: + return limits, None + counts: dict[str, int] = {} + for row in rows: + trajectory_id = str(row["trajectory_id"]) + counts[trajectory_id] = counts.get(trajectory_id, 0) + 1 + total_budget = base_limit * len(rows) + floor = max(256, (base_limit * 3) // 4) + for row in rows: + if counts[str(row["trajectory_id"])] > 1: + limits[int(row["state_id"])] = floor + used = sum( + min(len(str(row["text"])), limits[int(row["state_id"])]) + for row in rows + ) + bank = max(0, total_budget - used) + raised: list[dict[str, int]] = [] + for row in rows: + if bank <= 0: + break + if counts[str(row["trajectory_id"])] != 1: + continue + state_id = int(row["state_id"]) + current = limits[state_id] + ceiling = min(_MAX_QUERY_TEXT_CHARS, base_limit * 2) + desired = min(len(str(row["text"])), ceiling) + extra = min(bank, max(0, desired - current)) + if extra: + limits[state_id] += extra + bank -= extra + raised.append({"state_id": state_id, "chars": limits[state_id]}) + return limits, { + "base_char_limit": base_limit, + "total_char_budget": total_budget, + "raised_only_hits": raised, + } + + @staticmethod + def _rendered_hit_text(hit: TrajectoryHit) -> str: + """Mirror the official adapter's text rendering for token budgeting.""" + lines = [ + f"[{hit.exact_ref}]", + f"Trajectory: {hit.trajectory_id}", + f"Goal: {hit.goal}", + f"Outcome: {hit.outcome or ''}", + ( + f"State: {hit.state_index} " + f"(sequence {hit.sequence_ordinal}, step {hit.step})" + ), + f"URL: {hit.url}", + f"Incoming action: {hit.incoming_action or ''}", + ] + if hit.thoughts: + lines.append(f"Thought: {hit.thoughts}") + label = ( + f"Visible state excerpt (offset {hit.text_offset})" + if hit.text_truncated else "Visible state" + ) + lines.append(f"{label}: {hit.text}") + if hit.observed_at is not None: + lines.append( + f"Observed at: {hit.observed_at} (source: {hit.observed_at_source})" + ) + if hit.occurred_at is not None: + lines.append( + f"Occurred at: {hit.occurred_at} (source: {hit.occurred_at_source})" + ) + return "\n".join(lines) + + @staticmethod + def _trim_hit_text(hit: TrajectoryHit, target_tokens: int) -> TrajectoryHit: + from .tokens import count_tokens + + if count_tokens(hit.text) <= target_tokens: + return hit + low, high = 0, len(hit.text) + while low < high: + middle = (low + high + 1) // 2 + if count_tokens(hit.text[:middle]) <= target_tokens: + low = middle + else: + high = middle - 1 + keep = max(0, low) + start = max(0, (len(hit.text) - keep) // 2) + return replace( + hit, + text=hit.text[start:start + keep], + text_offset=hit.text_offset + start, + text_truncated=True, + ) + + @staticmethod + def _apply_sharp_token_budget( + hits: Sequence[TrajectoryHit], + token_budget: int, + ) -> tuple[list[TrajectoryHit], dict[str, Any]]: + from .tokens import count_tokens + + working = list(hits) + original_tokens = sum( + count_tokens(TrajectoryStore._rendered_hit_text(hit)) for hit in working + ) + dropped: list[str] = [] + while working: + fixed_tokens = sum( + count_tokens( + TrajectoryStore._rendered_hit_text(replace(hit, text="")) + ) + for hit in working + ) + if fixed_tokens <= token_budget: + break + dropped.append(working.pop().exact_ref) + if working: + fixed_tokens = sum( + count_tokens( + TrajectoryStore._rendered_hit_text(replace(hit, text="")) + ) + for hit in working + ) + available = max(0, token_budget - fixed_tokens) + per_hit = max(0, available // len(working)) + working = [ + TrajectoryStore._trim_hit_text(hit, per_hit) for hit in working + ] + final_tokens = sum( + count_tokens(TrajectoryStore._rendered_hit_text(hit)) for hit in working + ) + while working and final_tokens > token_budget: + largest_index = max( + range(len(working)), + key=lambda index: count_tokens(working[index].text), + ) + current = count_tokens(working[largest_index].text) + if current <= 0: + dropped.append(working.pop().exact_ref) + else: + working[largest_index] = TrajectoryStore._trim_hit_text( + working[largest_index], max(0, current - 8) + ) + final_tokens = sum( + count_tokens(TrajectoryStore._rendered_hit_text(hit)) + for hit in working + ) + return working, { + "text_token_budget": token_budget, + "rendered_text_tokens_before": original_tokens, + "rendered_text_tokens_after": final_tokens, + "dropped_evidence_refs": dropped, + } + + @staticmethod + def _select_diverse( + rows: Iterable[sqlite3.Row], + limit: int, + max_per_trajectory: int = 5, + ) -> list[sqlite3.Row]: selected: list[sqlite3.Row] = [] per_trajectory: dict[str, int] = {} for row in rows: trajectory_id = str(row["trajectory_id"]) - if per_trajectory.get(trajectory_id, 0) >= 5: + if per_trajectory.get(trajectory_id, 0) >= max_per_trajectory: continue selected.append(row) per_trajectory[trajectory_id] = per_trajectory.get(trajectory_id, 0) + 1 @@ -2052,6 +2551,7 @@ def _select_with_floor( global_rows: Sequence[sqlite3.Row], limit: int, floor_k: int, + max_per_trajectory: int = 5, ) -> list[sqlite3.Row]: """Policy A -- reserve ``floor_k`` nucleus slots for the top pure-BM25 states, then fill the remainder from the fused order. @@ -2061,7 +2561,11 @@ def _select_with_floor( trajectories monopolise the nucleus. Both the floor and the fill honour the same 5-per-trajectory diversity cap as ``_select_diverse``. """ - selected = list(TrajectoryStore._select_diverse(global_rows, floor_k)) + selected = list(TrajectoryStore._select_diverse( + global_rows, + floor_k, + max_per_trajectory=max_per_trajectory, + )) selected_ids = {int(row["state_id"]) for row in selected} per_trajectory: dict[str, int] = {} for row in selected: @@ -2074,7 +2578,7 @@ def _select_with_floor( if state_id in selected_ids: continue trajectory_id = str(row["trajectory_id"]) - if per_trajectory.get(trajectory_id, 0) >= 5: + if per_trajectory.get(trajectory_id, 0) >= max_per_trajectory: continue selected.append(row) selected_ids.add(state_id) @@ -2282,6 +2786,9 @@ def query( adjacency_radius: int = 0, adjacency_quota: int = 0, state_semantic_quota: int = 0, + diversity_cap: int = 0, + adaptive_excerpt: bool = False, + sharp_token_budget: int = 0, ) -> tuple[TrajectoryHit, ...]: if self.status != "complete": raise CorpusIdentityError("trajectory corpus must be finalized before query") @@ -2292,6 +2799,12 @@ def query( adjacency_radius = min(max(0, int(adjacency_radius)), _MAX_ADJACENCY_RADIUS) adjacency_quota = min(max(0, int(adjacency_quota)), _MAX_CANDIDATES) state_semantic_quota = min(max(0, int(state_semantic_quota)), _MAX_CANDIDATES) + diversity_cap = min(max(0, int(diversity_cap)), _MAX_DIVERSITY_CAP) + sharp_token_budget = min( + max(0, int(sharp_token_budget)), + _MAX_SHARP_TOKEN_BUDGET, + ) + adaptive_excerpt = bool(adaptive_excerpt) text_char_limit = min( max(256, int(text_char_limit)), _MAX_QUERY_TEXT_CHARS, @@ -2305,7 +2818,13 @@ def query( "delivered_evidence_refs": [], } return () - global_rows = self._fts_rows(expression, candidate_limit) + sharp_telemetry: dict[str, Any] | None = None + if sharp_token_budget > 0: + global_rows, sharp_telemetry = self._sharp_fts_rows( + query, candidate_limit + ) + else: + global_rows = self._fts_rows(expression, candidate_limit) semantic_ranks: list[tuple[int, float]] = [] semantic_attempt: TrajectorySemanticAttempt | None = None attempt_started = time.monotonic() @@ -2483,8 +3002,35 @@ def query( }) rows = expanded + question_template: str | None = None + if sharp_token_budget > 0: + question_template = self._question_template(query) + rows = self._template_order(rows, query, question_template) + + diversity_telemetry: dict[str, Any] | None = None + diversity_survivors: set[int] | None = None + if diversity_cap > 0: + # C1 is deliberately applied once, after lexical, source-semantic, + # adjacency, and state-semantic arms have composed. Filtering each + # arm independently would allow a capped hub to re-enter through a + # different arm. + rows, diversity_telemetry = self._cap_composed_pool( + rows, diversity_cap + ) + diversity_survivors = { + int(row["state_id"]) for row in rows + } + adjacent_reserve = min(6, limit // 3) if include_adjacent else 0 nucleus_limit = max(1, limit - adjacent_reserve) + lexical_candidates = ( + [ + row for row in global_rows + if int(row["state_id"]) in diversity_survivors + ] + if diversity_survivors is not None else global_rows + ) + per_trajectory = diversity_cap or 5 if arm_quota is not None: # Policy D (candidate-composition repair, issue #127): round-robin a # pure-lexical arm and the semantic/fused arm into the nucleus by the @@ -2496,8 +3042,16 @@ def query( # (``lexical_floor == 0`` reproduces the pure Policy D bytes). q_lex = max(0, int(arm_quota[0])) q_sem = max(0, int(arm_quota[1])) - arm_lex = self._select_diverse(global_rows, nucleus_limit) - arm_sem = self._select_diverse(rows, nucleus_limit) + arm_lex = self._select_diverse( + lexical_candidates, + nucleus_limit, + max_per_trajectory=per_trajectory, + ) + arm_sem = self._select_diverse( + rows, + nucleus_limit, + max_per_trajectory=per_trajectory, + ) selected = self._merge_arms( arm_lex, arm_sem, nucleus_limit, q_lex, q_sem, floor_k=lexical_floor, @@ -2508,10 +3062,18 @@ def query( # the rest. ``lexical_floor == 0`` (default) is byte-identical to the # historical fused-only selection below. selected = self._select_with_floor( - rows, global_rows, nucleus_limit, lexical_floor + rows, + lexical_candidates, + nucleus_limit, + lexical_floor, + max_per_trajectory=per_trajectory, ) else: - selected = self._select_diverse(rows, nucleus_limit) + selected = self._select_diverse( + rows, + nucleus_limit, + max_per_trajectory=diversity_cap or 5, + ) selected_ids = {int(row["state_id"]) for row in selected} match_kind_by_id = { int(row["state_id"]): candidate_kind.get(int(row["state_id"]), "fts") @@ -2524,6 +3086,12 @@ def query( if include_adjacent and selected and len(selected) < limit: nucleus_rows = list(selected) + selected_per_trajectory: dict[str, int] = {} + for row in selected: + trajectory_id = str(row["trajectory_id"]) + selected_per_trajectory[trajectory_id] = ( + selected_per_trajectory.get(trajectory_id, 0) + 1 + ) adjacent_by_nucleus: list[list[sqlite3.Row]] = [] for nucleus in nucleus_rows: adjacent_rows = self._conn.execute( @@ -2552,8 +3120,18 @@ def query( state_id = int(row["state_id"]) if state_id in selected_ids: continue + trajectory_id = str(row["trajectory_id"]) + if ( + diversity_cap > 0 + and selected_per_trajectory.get(trajectory_id, 0) + >= diversity_cap + ): + continue selected.append(row) selected_ids.add(state_id) + selected_per_trajectory[trajectory_id] = ( + selected_per_trajectory.get(trajectory_id, 0) + 1 + ) match_kind_by_id[state_id] = "adjacent" score_by_id[state_id] = score_by_id[int(nucleus["state_id"])] + 0.000001 made_progress = True @@ -2563,6 +3141,15 @@ def query( if not made_progress: break + adaptive_active = ( + adaptive_excerpt + and str(self.identity_payload.get("domain", "")).casefold() == "web" + ) + excerpt_limits, excerpt_telemetry = self._adaptive_excerpt_limits( + selected[:limit], + text_char_limit, + adaptive_active, + ) hits: list[TrajectoryHit] = [] for index, row in enumerate(selected[:limit]): state_id = int(row["state_id"]) @@ -2572,8 +3159,14 @@ def query( match_kind=match_kind_by_id[state_id], include_image=index < image_limit, query=query, - text_char_limit=text_char_limit, + text_char_limit=excerpt_limits[state_id], + dense_excerpt=adaptive_active, )) + budget_telemetry: dict[str, Any] | None = None + if sharp_token_budget > 0: + hits, budget_telemetry = self._apply_sharp_token_budget( + hits, sharp_token_budget + ) # Side-channel per-query telemetry (does not affect the returned hits). self._last_query_telemetry = { @@ -2596,6 +3189,19 @@ def query( ], "delivered_evidence_refs": [hit.exact_ref for hit in hits], } + if diversity_telemetry is not None: + # Present only when C1 is active; the complete survivor ids let the + # provider-free replay harness measure exact pool coverage without + # relying on the historical 64-row telemetry display bound. + self._last_query_telemetry["diversity_cap"] = diversity_telemetry + if excerpt_telemetry is not None: + self._last_query_telemetry["adaptive_excerpt"] = excerpt_telemetry + if sharp_telemetry is not None: + self._last_query_telemetry["sharp_compilation"] = { + **sharp_telemetry, + **(budget_telemetry or {}), + "question_template": question_template, + } if adjacency_radius > 0 and adjacency_quota > 0: # Present only when the H5(b) knob is active so the default # telemetry payload stays byte-identical (golden 451/451). From e99f342510a8a0fc0977bfc4d4c84d3e5b1fc3d6 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Fri, 24 Jul 2026 19:51:19 +0700 Subject: [PATCH 19/54] trajectory: default-off anti-boilerplate MMR (G) + title-boost (H) retrieval knobs (#143) Knob G (HERMES_LCM_ANTIBOILERPLATE, default-off): inside C1's per-trajectory MMR survivor selection, penalize a candidate by its mean lexical similarity to the other pooled states of its own trajectory and reward query-term density, so a trajectory's seats go to query-relevant states not repeated task headers. Knob H (HERMES_LCM_TITLE_BOOST, default-off): at the lexical candidate stage, stable-reorder states whose title/heading/field-label text contains an exact (case/punct-normalized) 2-4 gram question phrase ahead of same-band peers. Both deterministic, no model calls, omitted from query() when off. Golden 451/451 byte-identical on base and W3a stores; full-suite failure-set parity (+3 new passing tests, identical 60 pre-existing bad node ids); ruff clean. --- benchmarking/w3b_it4_replay.py | 188 +++++++++++++++++++ tests/test_trajectory_compact_delivery.py | 150 ++++++++++++++++ trajectory_store.py | 209 +++++++++++++++++++++- 3 files changed, 543 insertions(+), 4 deletions(-) create mode 100644 benchmarking/w3b_it4_replay.py diff --git a/benchmarking/w3b_it4_replay.py b/benchmarking/w3b_it4_replay.py new file mode 100644 index 000000000..6c6fc9888 --- /dev/null +++ b/benchmarking/w3b_it4_replay.py @@ -0,0 +1,188 @@ +#!/usr/bin/env python3 +"""Provider-free W3b iteration-4 replay for Knob G and Knob H. + +Reuses the frozen H5 target/preservation contract and the W3b C1 replay +infrastructure (:mod:`benchmarking.w3b_diversity_replay`). It re-runs only the +two grid cells named by the iteration-4 dispatch -- ``cap2`` (base replay +stores) and ``q32xcap2`` (W3a per-state-embedded working copies) -- three ways +each: the default C1 cell, the same cell with Knob G +(``antiboilerplate=True``), and with Knob H (``title_boost=True``). The delta +vs the frozen W3b build-report table is what the arm dispatch consumes. + +No embedding or reader provider is called. The state-semantic query direction +is reconstructed deterministically from frozen source-vector projections, the +same as the parent harness. This is provider-free component evidence, not an +official reader/judge/promotion result. +""" +from __future__ import annotations + +import argparse +import json +import time +from pathlib import Path +from typing import Any + +import sys + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from benchmarking.h3_composition_replay import ( # noqa: E402 + _H1, + _H31, + _RUN_ROOT, + ReplayContext, + golden_gate, +) +from benchmarking.h5_recall_replay import ( # noqa: E402 + _H5_TARGETS, + _LOSS_8, + H5Context, +) +from benchmarking.w3b_diversity_replay import ( # noqa: E402 + _DEFAULT_DB_DIR, + ProjectedStateReplayContext, + _evaluate_cell, +) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--run-root", type=Path, default=_RUN_ROOT) + parser.add_argument("--h1-artifacts", type=Path, default=_H1) + parser.add_argument("--h31-artifacts", type=Path, default=_H31) + parser.add_argument("--targets", type=Path, default=_H5_TARGETS) + parser.add_argument("--db-dir", type=Path, default=_DEFAULT_DB_DIR) + parser.add_argument("--out", type=Path, required=True) + parser.add_argument("--cap", type=int, default=2) + parser.add_argument("--state-quota", type=int, default=32) + args = parser.parse_args() + + args.out.parent.mkdir(parents=True, exist_ok=True) + base_context = ReplayContext( + args.run_root, args.h1_artifacts, args.h31_artifacts + ) + state_context = ProjectedStateReplayContext( + args.run_root, args.h1_artifacts, args.h31_artifacts, args.db_dir + ) + base_h5 = H5Context(base_context, args.targets) + state_h5 = H5Context(state_context, args.targets) + + print("== GOLDEN GATES (all W3b knobs off, incl. G/H) ==", flush=True) + base_golden = golden_gate(base_context) + state_golden = golden_gate(state_context) + print( + f" base {base_golden['passed']}/{base_golden['total']} | " + f"state-db {state_golden['passed']}/{state_golden['total']}", + flush=True, + ) + if ( + base_golden["passed"] != base_golden["total"] + or state_golden["passed"] != state_golden["total"] + ): + print("GOLDEN GATE FAILED -- aborting before cells", flush=True) + return 1 + + reconciliation = json.loads( + ( + args.h1_artifacts / "old-rescore" / "reconciliation_sets.json" + ).read_text() + ) + preserved_qids = sorted({ + item.split("/", 1)[1] for item in reconciliation["preserved"] + }) + all_qids = sorted({ + *preserved_qids, + *(str(case["qid"]) for case in base_h5.cases), + *(qid for _domain, qid in _LOSS_8), + }) + base_delivered = {qid: base_context.deliver(qid) for qid in all_qids} + state_base_delivered = { + qid: state_context.deliver(qid) for qid in all_qids + } + if base_delivered != state_base_delivered: + print("STATE-DB BASELINE DIFFERS -- aborting cells", flush=True) + return 2 + + cap = args.cap + quota = args.state_quota + cells: list[dict[str, Any]] = [] + + def run(context, h5, label, kwargs, mode): + started = time.perf_counter() + row = _evaluate_cell(context, h5, kwargs, base_delivered, preserved_qids) + row["label"] = label + row["state_query_mode"] = mode + row["elapsed_s"] = round(time.perf_counter() - started, 3) + cells.append(row) + print( + f" {label}: pool {row['target_pool_covered']}/30 " + f"(new {row['new_pool_recovery']}) | " + f"dRec@16 {row['delivered_recall_at_16']}/30 | " + f"pres {row['preservation_retained']}/154", + flush=True, + ) + + # cap2 family (base replay stores) + run(base_context, base_h5, f"cap{cap}", {"diversity_cap": cap}, "not_used") + run( + base_context, base_h5, f"cap{cap}+G", + {"diversity_cap": cap, "antiboilerplate": True}, "not_used", + ) + run( + base_context, base_h5, f"cap{cap}+H", + {"diversity_cap": cap, "title_boost": True}, "not_used", + ) + + # q32xcap2 family (W3a per-state-embedded working copies) + run( + state_context, state_h5, f"q{quota}xcap{cap}", + {"state_semantic_quota": quota, "diversity_cap": cap}, + "recorded-source-projection", + ) + run( + state_context, state_h5, f"q{quota}xcap{cap}+G", + { + "state_semantic_quota": quota, + "diversity_cap": cap, + "antiboilerplate": True, + }, + "recorded-source-projection", + ) + run( + state_context, state_h5, f"q{quota}xcap{cap}+H", + { + "state_semantic_quota": quota, + "diversity_cap": cap, + "title_boost": True, + }, + "recorded-source-projection", + ) + + payload = { + "proof_boundary": ( + "Provider-free component replay only; reconstructed query directions " + "do not prove official reader/judge accuracy or promotion." + ), + "golden_gate": {"base": base_golden, "state_db": state_golden}, + "targets_manifest": str(args.targets), + "db_dir": str(args.db_dir), + "cap": cap, + "state_quota": quota, + "preserved_universe": len(preserved_qids), + "external_provider_calls": sum( + provider.external_calls + for provider in state_context.providers.values() + ), + "query_projections_built": { + domain: provider.projections_built + for domain, provider in state_context.providers.items() + }, + "cells": cells, + } + args.out.write_text(json.dumps(payload, indent=2) + "\n") + print(f"wrote {args.out}", flush=True) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_trajectory_compact_delivery.py b/tests/test_trajectory_compact_delivery.py index 2dd1dc3f6..9b5eb40fb 100644 --- a/tests/test_trajectory_compact_delivery.py +++ b/tests/test_trajectory_compact_delivery.py @@ -141,12 +141,15 @@ def test_all_w3b_knobs_default_off_preserve_bytes_and_telemetry(tmp_path: Path): diversity_cap=0, adaptive_excerpt=False, sharp_token_budget=0, + antiboilerplate=False, + title_boost=False, ) assert [hit.to_dict() for hit in explicit_off] == baseline_payload assert store.last_query_telemetry() == baseline_telemetry assert "diversity_cap" not in baseline_telemetry assert "adaptive_excerpt" not in baseline_telemetry assert "sharp_compilation" not in baseline_telemetry + assert "title_boost" not in baseline_telemetry def test_c1_caps_after_state_arm_composition_and_reports_trajectory(tmp_path: Path): @@ -249,3 +252,150 @@ def test_c3_typed_queries_exact_title_template_and_budget(tmp_path: Path): } assert sharp["exact_title_boosts"] assert sharp["rendered_text_tokens_after"] <= budget + + +def _cap_row( + state_id: int, + trajectory_id: str, + goal: str, + text: str, +) -> dict[str, object]: + return { + "state_id": state_id, + "trajectory_id": trajectory_id, + "goal": goal, + "url": "", + "incoming_action": None, + "text": text, + } + + +def _t1_selected(cap_telemetry: dict) -> list[int]: + entry = next( + item + for item in cap_telemetry["trajectories"] + if item["trajectory_id"] == "t1" + ) + return list(entry["selected_state_ids"]) + + +def test_g_antiboilerplate_reweights_c1_survivor_selection(): + # One trajectory ("t1") with two near-identical task-header boilerplate + # states ranked ahead of a distinct, query-relevant needle state. Padding + # states from a second trajectory shrink the per-position relevance gap so + # the boilerplate/density signals -- not raw rank -- decide the survivor. + rows = [ + _cap_row( + 1, "t1", "export dashboard", + "export dashboard task header repeated boilerplate navigation", + ), + _cap_row( + 2, "t1", "export dashboard", + "export dashboard task header repeated boilerplate navigation footer", + ), + _cap_row( + 3, "t1", "export dashboard", + "quarterly revenue figure answer distinct value column", + ), + ] + rows += [ + _cap_row(100 + index, "pad", "unrelated pad", f"pad body token {index}") + for index in range(12) + ] + query_terms = TrajectoryStore._query_term_set( + "quarterly revenue figure answer column" + ) + + off_rows, off_telemetry = TrajectoryStore._cap_composed_pool(rows, 1) + assert _t1_selected(off_telemetry) == [1] + assert "antiboilerplate" not in off_telemetry + + # Default-off keyword form is byte-identical to the positional default. + off_rows_kw, off_telemetry_kw = TrajectoryStore._cap_composed_pool( + rows, 1, antiboilerplate=False, query_terms=query_terms + ) + assert [int(row["state_id"]) for row in off_rows_kw] == [ + int(row["state_id"]) for row in off_rows + ] + assert off_telemetry_kw == off_telemetry + + on_rows, on_telemetry = TrajectoryStore._cap_composed_pool( + rows, 1, antiboilerplate=True, query_terms=query_terms + ) + assert _t1_selected(on_telemetry) == [3] + assert 3 in {int(row["state_id"]) for row in on_rows} + + scored = { + entry["state_id"]: entry + for entry in on_telemetry["antiboilerplate"]["scored"] + } + # Needle has the query-term density; boilerplate siblings resemble each other. + assert scored[3]["density"] > scored[1]["density"] + assert scored[1]["boilerplate"] > scored[3]["boilerplate"] + assert scored[2]["boilerplate"] > scored[3]["boilerplate"] + + +def test_h_title_boost_promotes_exact_ngram_matches(): + rows = [ + { + "state_id": 1, + "trajectory_id": "a", + "goal": "generic dashboard", + "url": "https://example.test/a/0", + "text": "some unrelated body text about panels and widgets", + }, + { + "state_id": 2, + "trajectory_id": "b", + "goal": "products grid", + "url": "https://example.test/b/0", + "text": "column header Last Updated At value 2024-01-02", + }, + { + "state_id": 3, + "trajectory_id": "c", + "goal": "orders", + "url": "https://example.test/c/0", + "text": "purchase date column with totals", + }, + ] + reordered, telemetry = TrajectoryStore._apply_title_boost( + rows, "What is the Last Updated At column value?" + ) + assert [int(row["state_id"]) for row in reordered][0] == 2 + assert telemetry["boosted_count"] == 1 + assert telemetry["boosted"][0]["state_id"] == 2 + assert "last updated at" in telemetry["boosted"][0]["phrases"] + + # No matching phrase leaves the pool order byte-identical. + same, empty = TrajectoryStore._apply_title_boost( + rows, "completely orthogonal unrelated inquiry" + ) + assert [int(row["state_id"]) for row in same] == [ + int(row["state_id"]) for row in rows + ] + assert empty["boosted_count"] == 0 + + +def test_h_title_boost_query_path_emits_telemetry(tmp_path: Path): + store = _build_store(tmp_path) + query = "widget export duplicate description field" + + off = store.query(query, image_limit=0, include_adjacent=False) + off_payload = [hit.to_dict() for hit in off] + assert "title_boost" not in store.last_query_telemetry() + + on = store.query( + query, image_limit=0, include_adjacent=False, title_boost=True + ) + telemetry = store.last_query_telemetry()["title_boost"] + assert telemetry["boosted_count"] >= 1 + assert any( + "duplicate description field" in " ".join(entry["phrases"]) + for entry in telemetry["boosted"] + ) + # The exact-phrase state is delivered under the boost. + assert any("alpha-answer" in hit.text for hit in on) + # Default-off path is unaffected by adding the knob at call time. + reconfirm = store.query(query, image_limit=0, include_adjacent=False) + assert [hit.to_dict() for hit in reconfirm] == off_payload diff --git a/trajectory_store.py b/trajectory_store.py index ba790e61c..163694a8e 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -55,6 +55,21 @@ _MAX_ADJACENCY_RADIUS = 8 _MAX_DIVERSITY_CAP = 24 _MAX_SHARP_TOKEN_BUDGET = 4_000 +# Knob G (HERMES_LCM_ANTIBOILERPLATE): additive re-weighting inside the C1 +# per-trajectory MMR survivor selection. A candidate is penalized by how much +# it lexically resembles the OTHER pooled states of its own trajectory +# (boilerplate looks like its siblings) and rewarded for query-term density, so +# a trajectory's allotted seats go to query-relevant states rather than repeated +# task headers / page furniture. Both terms are in [0, 1]; the weights keep the +# base position-relevance signal dominant while giving the two signals real +# re-ranking influence. +_ANTIBOILERPLATE_BOILERPLATE_WEIGHT = 0.30 +_ANTIBOILERPLATE_DENSITY_WEIGHT = 0.30 +# Knob H (HERMES_LCM_TITLE_BOOST): contiguous question n-gram sizes matched +# against a candidate's normalized title/heading/field-label text at the lexical +# candidate stage. +_TITLE_BOOST_MIN_GRAM = 2 +_TITLE_BOOST_MAX_GRAM = 4 _MAX_SOURCE_JSON_CHARS = 16_000_000 _MAX_TEXT_CHARS = 2_000_000 _MAX_SEMANTIC_DOCUMENT_CHARS = 48_000 @@ -2124,10 +2139,111 @@ def _candidate_term_set(row: Any) -> set[str]: if len(token) >= 3 and token not in _STOPWORDS } + @staticmethod + def _query_term_set(query: str) -> frozenset[str]: + """Tokenize a query into the same lexical signature space as states. + + Mirrors ``_candidate_term_set`` tokenization so query-term density is + measured against the identical vocabulary the MMR signatures use. + """ + return frozenset( + token + for token in re.findall(r"[a-z0-9][a-z0-9_-]+", query.casefold()) + if len(token) >= 3 and token not in _STOPWORDS + ) + + @staticmethod + def _title_field_text(row: Any) -> str: + """Normalized title/heading/field-label text for the Knob H boost. + + Punctuation folds to single spaces and the result is padded with a + leading/trailing space so a matched n-gram is checked at word + boundaries via a plain substring test. + """ + raw = " ".join( + str(row[key] or "") + for key in ("goal", "url", "text") + )[:12_000].casefold() + collapsed = re.sub(r"[^a-z0-9]+", " ", raw) + return f" {' '.join(collapsed.split())} " + + @staticmethod + def _query_title_ngrams(query: str) -> list[str]: + """Contiguous 2-4 gram question phrases for the Knob H title boost. + + Grams composed entirely of stopwords are dropped (a pure ``what is + the`` gram carries no signal); the remaining grams are normalized the + same way as ``_title_field_text`` and returned in stable first-seen + order. + """ + words = [ + word.casefold() + for word in re.findall(r"[A-Za-z0-9][A-Za-z0-9_-]*", query) + ] + grams: list[str] = [] + seen: set[str] = set() + for size in range(_TITLE_BOOST_MIN_GRAM, _TITLE_BOOST_MAX_GRAM + 1): + for start in range(0, max(0, len(words) - size + 1)): + window = words[start:start + size] + if all(token in _STOPWORDS for token in window): + continue + normalized = re.sub(r"[^a-z0-9]+", " ", " ".join(window)).strip() + if not normalized or normalized in seen: + continue + seen.add(normalized) + grams.append(normalized) + return grams + + @classmethod + def _apply_title_boost( + cls, + rows: Sequence[Any], + query: str, + ) -> tuple[list[Any], dict[str, Any]]: + """Knob H: stable-reorder the lexical pool by exact title n-gram hits. + + A candidate is boosted by the number of DISTINCT contiguous question + 2-4 grams that appear (case/punctuation-normalized) in its + title/heading/field-label text. The reorder is stable: candidates keep + their original relative order within an equal-boost band, so with no + matches the pool is byte-identical to the input. Pure-lexical and + deterministic -- no model calls. + """ + grams = cls._query_title_ngrams(query) + boosts: dict[int, int] = {} + matched: list[dict[str, Any]] = [] + if grams: + for row in rows: + field_text = cls._title_field_text(row) + hits = [gram for gram in grams if f" {gram} " in field_text] + if hits: + state_id = int(row["state_id"]) + boosts[state_id] = len(hits) + matched.append({"state_id": state_id, "phrases": hits}) + reordered = [ + row + for _key, row in sorted( + enumerate(rows), + key=lambda item: ( + -boosts.get(int(item[1]["state_id"]), 0), + item[0], + ), + ) + ] + telemetry = { + "ngrams": grams, + "boosted": matched, + "boosted_count": len(matched), + } + return reordered, telemetry + @staticmethod def _cap_composed_pool( rows: Sequence[Any], cap: int, + *, + antiboilerplate: bool = False, + query_terms: frozenset[str] | None = None, ) -> tuple[list[Any], dict[str, Any]]: """Apply one global per-trajectory cap after all candidate arms compose. @@ -2135,7 +2251,16 @@ def _cap_composed_pool( MMR-style relevance/redundancy score. The returned pool preserves the original cross-trajectory order, so the cap changes only which states survive, not unrelated tie-breaking. + + When ``antiboilerplate`` is set (Knob G, default-off) the per-candidate + MMR score is additionally penalized by the candidate's mean lexical + similarity to the OTHER pooled states of its own trajectory and rewarded + by its query-term density, so a trajectory's seats go to query-relevant + states rather than repeated task-header boilerplate. ``query_terms`` is + the tokenized query used for the density reward; both signals are inert + when ``antiboilerplate`` is ``False`` (byte-identical to the base cap). """ + query_terms = query_terms or frozenset() unique_rows: list[Any] = [] seen_state_ids: set[int] = set() for row in rows: @@ -2149,6 +2274,7 @@ def _cap_composed_pool( grouped.setdefault(str(row["trajectory_id"]), []).append((position, row)) survivor_ids: set[int] = set() details: list[dict[str, Any]] = [] + antiboilerplate_scores: list[dict[str, Any]] = [] for trajectory_id, candidates in grouped.items(): if len(candidates) <= cap: survivor_ids.update(int(row["state_id"]) for _position, row in candidates) @@ -2157,6 +2283,31 @@ def _cap_composed_pool( int(row["state_id"]): TrajectoryStore._candidate_term_set(row) for _position, row in candidates } + boilerplate: dict[int, float] = {} + density: dict[int, float] = {} + if antiboilerplate: + for _position, row in candidates: + state_id = int(row["state_id"]) + signature = signatures[state_id] + sibling_scores: list[float] = [] + for _other_position, other_row in candidates: + other_id = int(other_row["state_id"]) + if other_id == state_id: + continue + other_signature = signatures[other_id] + union = signature | other_signature + sibling_scores.append( + len(signature & other_signature) / len(union) + if union else 0.0 + ) + boilerplate[state_id] = ( + sum(sibling_scores) / len(sibling_scores) + if sibling_scores else 0.0 + ) + density[state_id] = ( + len(signature & query_terms) / len(signature) + if signature else 0.0 + ) selected: list[tuple[int, Any]] = [] remaining = list(candidates) while remaining and len(selected) < cap: @@ -2167,7 +2318,8 @@ def _cap_composed_pool( relevance = 1.0 - ( original_position / max(1, len(unique_rows) - 1) ) - signature = signatures[int(row["state_id"])] + state_id = int(row["state_id"]) + signature = signatures[state_id] redundancy = 0.0 for _selected_position, selected_row in selected: selected_signature = signatures[int(selected_row["state_id"])] @@ -2178,13 +2330,32 @@ def _cap_composed_pool( ) redundancy = max(redundancy, similarity) mmr_score = (0.72 * relevance) - (0.28 * redundancy) - key = (mmr_score, -original_position, -int(row["state_id"])) + if antiboilerplate: + mmr_score += ( + _ANTIBOILERPLATE_DENSITY_WEIGHT * density[state_id] + ) - ( + _ANTIBOILERPLATE_BOILERPLATE_WEIGHT + * boilerplate[state_id] + ) + key = (mmr_score, -original_position, -state_id) if best_key is None or key > best_key: best_key = key best_item = item assert best_item is not None selected.append(best_item) remaining.remove(best_item) + if antiboilerplate: + antiboilerplate_scores.extend( + { + "state_id": int(row["state_id"]), + "boilerplate": round(boilerplate[int(row["state_id"])], 6), + "density": round(density[int(row["state_id"])], 6), + "selected": int(row["state_id"]) in { + int(sel_row["state_id"]) for _pos, sel_row in selected + }, + } + for _position, row in candidates + ) selected_ids = [int(row["state_id"]) for _position, row in selected] removed_ids = [ int(row["state_id"]) @@ -2203,7 +2374,7 @@ def _cap_composed_pool( filtered = [ row for row in unique_rows if int(row["state_id"]) in survivor_ids ] - return filtered, { + telemetry: dict[str, Any] = { "cap": cap, "pool_before": len(unique_rows), "pool_after": len(filtered), @@ -2211,6 +2382,13 @@ def _cap_composed_pool( "trajectories": details, "survivor_state_ids": [int(row["state_id"]) for row in filtered], } + if antiboilerplate: + telemetry["antiboilerplate"] = { + "density_weight": _ANTIBOILERPLATE_DENSITY_WEIGHT, + "boilerplate_weight": _ANTIBOILERPLATE_BOILERPLATE_WEIGHT, + "scored": antiboilerplate_scores, + } + return filtered, telemetry @staticmethod def _exact_query_phrases(query: str) -> list[str]: @@ -2789,6 +2967,8 @@ def query( diversity_cap: int = 0, adaptive_excerpt: bool = False, sharp_token_budget: int = 0, + antiboilerplate: bool = False, + title_boost: bool = False, ) -> tuple[TrajectoryHit, ...]: if self.status != "complete": raise CorpusIdentityError("trajectory corpus must be finalized before query") @@ -2805,6 +2985,8 @@ def query( _MAX_SHARP_TOKEN_BUDGET, ) adaptive_excerpt = bool(adaptive_excerpt) + antiboilerplate = bool(antiboilerplate) + title_boost = bool(title_boost) text_char_limit = min( max(256, int(text_char_limit)), _MAX_QUERY_TEXT_CHARS, @@ -2825,6 +3007,16 @@ def query( ) else: global_rows = self._fts_rows(expression, candidate_limit) + # Knob H (title boost, default-off): stable-reorder the lexical + # candidate pool so states whose title/heading/field-label text exactly + # contains a 2-4 gram question phrase rank ahead of same-band peers. + # ``title_boost is False`` (default) leaves ``global_rows`` untouched and + # reproduces current bytes. + title_boost_telemetry: dict[str, Any] | None = None + if title_boost: + global_rows, title_boost_telemetry = self._apply_title_boost( + global_rows, query + ) semantic_ranks: list[tuple[int, float]] = [] semantic_attempt: TrajectorySemanticAttempt | None = None attempt_started = time.monotonic() @@ -3015,7 +3207,12 @@ def query( # arm independently would allow a capped hub to re-enter through a # different arm. rows, diversity_telemetry = self._cap_composed_pool( - rows, diversity_cap + rows, + diversity_cap, + antiboilerplate=antiboilerplate, + query_terms=( + self._query_term_set(query) if antiboilerplate else None + ), ) diversity_survivors = { int(row["state_id"]) for row in rows @@ -3217,6 +3414,10 @@ def query( "quota": state_semantic_quota, "admitted": state_semantic_admitted, } + if title_boost_telemetry is not None: + # Knob H: present only when title boost is active so the default + # telemetry payload stays byte-identical (golden 451/451). + self._last_query_telemetry["title_boost"] = title_boost_telemetry return tuple(hits) def resolve_exact_ref(self, exact_ref: str) -> TrajectoryHit: From b2f228c4f401d8afa306d9e7ae2d31094105400c Mon Sep 17 00:00:00 2001 From: EVA Date: Wed, 29 Jul 2026 02:45:16 +0700 Subject: [PATCH 20/54] recall: reduce a raw question to FTS5 terms before MATCH (#168) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit FTS5 barewords accept only alphanumerics, so a natural-language question (`?`, apostrophes, commas, `&`, `$`) was a syntax error at MATCH. The query fell through to the LIKE full-scan, which blows recall_query_timeout_s at scale and returns EMPTY -- 100% of queries at 8,000+ sessions (FINDING-F31 §3). The sanitization belonged in the product, not in a benchmark harness that happened to know the trick. sanitize_fts5_query now maps every non-alphanumeric character outside a balanced phrase quote to a separator (the transform the Phase 1A A3 arm used), matching how the default unicode61 tokenizer splits the INDEXED text -- so no term is lost. LIKE stays the fallback for what FTS cannot express (CJK/emoji/compound tokens) and for a query with nothing left after sanitization; that path now scores punctuation-only queries on the raw text so its literal substring behavior is unchanged. --- dag.py | 10 +++++-- search_query.py | 26 ++++++++++++++---- store.py | 10 +++++-- tests/test_lcm_core.py | 62 ++++++++++++++++++++++++++++++++++++++++++ 4 files changed, 98 insertions(+), 10 deletions(-) diff --git a/dag.py b/dag.py index 3a935c04f..b20ad929a 100644 --- a/dag.py +++ b/dag.py @@ -564,7 +564,11 @@ def search(self, query: str, session_id: str | None = None, safe_query = sanitize_fts5_query(query) terms = extract_search_terms(safe_query) phrases = extract_quoted_phrases(safe_query) - if requires_like_fallback(query): + # LIKE is the fallback for text FTS cannot express (CJK/emoji/compound + # tokens) and for a query with no term left after sanitization. A raw + # natural-language question is NOT one of those: it sanitizes to a term + # form the index answers, so it stays on the FTS path (F31 §3). + if requires_like_fallback(query) or not safe_query: return self._search_like(query, session_id=session_id, limit=limit, sort=sort, source=source) order_by = _build_search_order_by(sort, "COALESCE(n.latest_at, n.created_at)") @@ -638,7 +642,9 @@ def search(self, query: str, session_id: str | None = None, def _search_like(self, query: str, session_id: str | None = None, limit: int = 20, sort: str | None = None, source: str | None = None) -> List[SummaryNode]: - safe_query = sanitize_fts5_query(query) + # A query that sanitizes away entirely (pure punctuation) still has a + # literal substring meaning here, so score it on the raw text. + safe_query = sanitize_fts5_query(query) or query terms = extract_search_terms(safe_query) phrases = extract_quoted_phrases(safe_query) if not terms: diff --git a/search_query.py b/search_query.py index 920dd8ad1..4eaedaa6c 100644 --- a/search_query.py +++ b/search_query.py @@ -26,16 +26,30 @@ _RISKY_FTS_TOKEN_RE = re.compile(r"[A-Za-z0-9][\-:/][A-Za-z0-9]") _SPLIT_PUNCT_RE = re.compile(r"[-:/]+") _STRIP_EDGE_PUNCT = "\"'()[]{}.,;" -# Characters that are special in FTS5 query syntax -_FTS5_SPECIAL_CHARS = frozenset('"()*^-:{}.') + + +def _fts5_safe_char(char: str) -> str: + """Map one unquoted character to its FTS5 bareword-safe form. + + An FTS5 bareword accepts only alphanumerics; every other character is + either query syntax (``"()*^-:{}.``), a string delimiter (``'``), or a + plain syntax error (``? , & $ ! % = < ;`` ...). A raw natural-language + question therefore fails ``MATCH`` outright and used to fall through to the + LIKE full-scan, which blows the recall deadline at scale and returns + nothing (F31 §3). Substituting a separator instead lets a question reach + the index in its term form; the default unicode61 tokenizer splits the + INDEXED text on exactly the same boundary, so no term is lost by the + substitution. + """ + return char if (char.isalnum() or char.isspace()) else " " def _sanitize_unquoted_fts5_fragment(text: str) -> str: - return "".join(" " if char in _FTS5_SPECIAL_CHARS else char for char in text) + return "".join(_fts5_safe_char(char) for char in text) def sanitize_fts5_query(query: str) -> str: - """Strip FTS5 syntax operators while preserving balanced phrase quotes.""" + """Reduce a query to FTS5-safe terms, preserving balanced phrase quotes.""" if not query: return "" @@ -59,10 +73,10 @@ def sanitize_fts5_query(query: str) -> str: if in_quote: quote_buffer.append(char) continue - result.append(" " if char in _FTS5_SPECIAL_CHARS else char) + result.append(_fts5_safe_char(char)) if in_quote and quote_buffer: result.extend(_sanitize_unquoted_fts5_fragment("".join(quote_buffer))) - return "".join(result).strip() + return " ".join("".join(result).split()) _WORD_RE = re.compile(r"[\w-]+", re.UNICODE) diff --git a/store.py b/store.py index 483290133..1d847d651 100644 --- a/store.py +++ b/store.py @@ -1112,7 +1112,11 @@ def search(self, query: str, session_id: str | None = None, safe_query = sanitize_fts5_query(query) terms = extract_search_terms(safe_query) phrases = extract_quoted_phrases(safe_query) - if requires_like_fallback(query): + # LIKE is the fallback for text FTS cannot express (CJK/emoji/compound + # tokens) and for a query with no term left after sanitization. A raw + # natural-language question is NOT one of those: it sanitizes to a term + # form the index answers, so it stays on the FTS path (F31 §3). + if requires_like_fallback(query) or not safe_query: return self._search_like( query, session_id=session_id, @@ -1228,7 +1232,9 @@ def _search_like(self, query: str, session_id: str | None = None, role: str | None = None, time_from: float | None = None, time_to: float | None = None) -> List[Dict[str, Any]]: - safe_query = sanitize_fts5_query(query) + # A query that sanitizes away entirely (pure punctuation) still has a + # literal substring meaning here, so score it on the raw text. + safe_query = sanitize_fts5_query(query) or query terms = extract_search_terms(safe_query) phrases = extract_quoted_phrases(safe_query) if not terms: diff --git a/tests/test_lcm_core.py b/tests/test_lcm_core.py index e2ea58b52..6cb3e06e8 100644 --- a/tests/test_lcm_core.py +++ b/tests/test_lcm_core.py @@ -1847,6 +1847,51 @@ def test_search_like_fallback_splits_unbalanced_quote_terms(self, store): assert len(results) == 1 assert results[0]["content"] == "foo bar baz" + def test_search_keeps_raw_question_on_the_fts_path(self, store): + """A natural-language question must not degrade to the LIKE full-scan. + + ``?`` and ``'`` are FTS5 syntax errors, so the question used to fail + MATCH and fall through to a LIKE scan that blows the recall deadline at + scale and returns nothing (F31 §3, issue #168). + """ + store.append("sess1", {"role": "user", "content": "my dog's vet appointment was tuesday"}) + store._search_like = lambda *args, **kwargs: pytest.fail( + "raw question fell back to the LIKE full-scan" + ) + + results = store.search("my dog's vet appointment?", session_id="sess1") + + assert len(results) == 1 + assert results[0]["content"] == "my dog's vet appointment was tuesday" + + def test_search_keeps_punctuated_question_on_the_fts_path(self, store): + store.append("sess1", {"role": "user", "content": "budget revenue q3 totals recorded"}) + store._search_like = lambda *args, **kwargs: pytest.fail( + "punctuated question fell back to the LIKE full-scan" + ) + + results = store.search("budget & revenue, q3 $ totals?", session_id="sess1") + + assert len(results) == 1 + assert results[0]["content"] == "budget revenue q3 totals recorded" + + def test_search_falls_back_to_like_when_query_sanitizes_empty(self, store): + store.append("sess1", {"role": "user", "content": "what??? really"}) + calls: list[str] = [] + original = store._search_like + + def _recording(query, **kwargs): + calls.append(query) + return original(query, **kwargs) + + store._search_like = _recording + + results = store.search("???", session_id="sess1") + + assert calls == ["???"] + assert len(results) == 1 + assert results[0]["content"] == "what??? really" + def test_search_uses_sanitized_terms_for_directness_scoring(self, store): store.append("sess1", {"role": "user", "content": "vendoring external support stays plugin-only"}) @@ -3165,6 +3210,23 @@ def test_sanitize_fts5_query_replaces_period_in_unquoted_terms(self): assert sanitize_fts5_query("api.v2") == "api v2" assert sanitize_fts5_query("hermes.lcm") == "hermes lcm" + def test_sanitize_fts5_query_reduces_a_question_to_terms(self): + # ? , & $ ' are all FTS5 syntax errors, not just the documented + # operators, so a raw question has to reach MATCH in term form (#168). + assert ( + sanitize_fts5_query("What did I say about my dog's vet appointment?") + == "What did I say about my dog s vet appointment" + ) + assert sanitize_fts5_query("budget & revenue, q3 $ totals") == "budget revenue q3 totals" + + def test_sanitize_fts5_query_leaves_clean_queries_unchanged(self): + assert sanitize_fts5_query("docker deploy notes") == "docker deploy notes" + assert sanitize_fts5_query("東京 memo") == "東京 memo" + + def test_sanitize_fts5_query_empties_a_punctuation_only_query(self): + assert sanitize_fts5_query("???") == "" + assert sanitize_fts5_query("!!! ***") == "" + def test_ensure_external_content_fts_skips_rebuild_when_disk_is_low(self, tmp_path, monkeypatch): conn = sqlite3.connect(tmp_path / "low-disk.db") conn.executescript( From f960d9f23ecded30fbafc00d616c18866a7303e8 Mon Sep 17 00:00:00 2001 From: EVA Date: Wed, 29 Jul 2026 02:53:58 +0700 Subject: [PATCH 21/54] recall: scan the whole vector corpus in batches, not a 25k recency window (#167) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit recall_scan_rows was a HARD bound on the brute-force scan, so at 185,175 vectors lcm_recall scored only the 25,000 most-recent and 86% of memory was invisible to semantic retrieval. All-gold recall@25 collapsed 0.82 -> 0.00 across the ladder exactly as gold content aged out of the window, and the no-deadline arm reproduced it: coverage, not time (FINDING-F31 §2). The "all conversations, all time" promise was false at scale. The scan now covers the WHOLE corpus in batches with a running top-k across them. recall_scan_rows becomes the BATCH SIZE, so it bounds vectors resident per batch (peak memory) instead of the corpus reached; the arithmetic is trivial either way (185k x 384-dim is ~0.14 GFLOPs). No ANN index here -- that is deferred until profiling shows the full scan too slow. lcm_grep is untouched: full_scan is off by default, and a single batch of bounded_scan_rows candidates is byte-for-byte the previous behavior. The degraded/degraded_reason machinery is preserved and now means what it says: coverage degrades to 'bounded' only when an explicit bound truncates the scan -- recall_scan_max_rows (0 = unlimited, for a pathological corpus) or recall_scan_budget_s (0 = no early stop). Both default to no early stop, so a default recall discloses nothing because nothing is hidden. --- config.py | 26 +++- retrieval_core.py | 23 +++- tests/test_lcm_recall.py | 63 +++++++++- tests/test_vector_store.py | 116 ++++++++++++++++++ tools.py | 19 +++ vector_store.py | 245 +++++++++++++++++++++++++++---------- 6 files changed, 413 insertions(+), 79 deletions(-) diff --git a/config.py b/config.py index f0b39c5a2..34fdc5643 100644 --- a/config.py +++ b/config.py @@ -372,6 +372,8 @@ class _EnvFieldSpec: _EnvFieldSpec("embeddings_enabled", "LCM_EMBEDDINGS_ENABLED", bool), _EnvFieldSpec("rerank_enabled", "LCM_RERANK_ENABLED", bool), _EnvFieldSpec("recall_scan_rows", "LCM_RECALL_SCAN_ROWS", int), + _EnvFieldSpec("recall_scan_max_rows", "LCM_RECALL_SCAN_MAX_ROWS", int), + _EnvFieldSpec("recall_scan_budget_s", "LCM_RECALL_SCAN_BUDGET_S", float), _EnvFieldSpec("proactive_recall_enabled", "LCM_PROACTIVE_RECALL_ENABLED", bool), _EnvFieldSpec("proactive_recall_min_score", "LCM_PROACTIVE_RECALL_MIN_SCORE", float), _EnvFieldSpec("proactive_recall_budget_tokens", "LCM_PROACTIVE_RECALL_BUDGET_TOKENS", int), @@ -630,12 +632,26 @@ class LCMConfig: # rescore) KNN: M = knn_prescreen_multiplier x k survivors are rescored. # Larger widens the approximate prescreen toward exact recall at more cost. knn_prescreen_multiplier: int = 4 - # lcm_recall candidate-scan bound. lcm_recall promises "all conversations, - # all time", so it must NOT inherit the small recency-truncating grep bound - # above (that structurally hides the oldest memories). This larger bound - # (still deadline-guarded) lets recall cover a realistic forever-memory - # corpus while capping worst-case cost on a very large one. + # lcm_recall candidate-scan BATCH SIZE. lcm_recall promises "all + # conversations, all time", so it must NOT inherit the small + # recency-truncating grep bound above (that structurally hides the oldest + # memories). It used to be a hard bound, which made the promise false at + # scale: at 185k vectors it scored only the 25k most-recent and recall for + # ageing content went to zero (FINDING-F31 §2). The scan now covers the + # WHOLE corpus and this value bounds only how many vectors are resident per + # batch (a running top-k spans the batches), so it trades peak memory, not + # coverage. recall_scan_rows: int = 25_000 + # Hard candidate cap for the lcm_recall scan; 0 = unlimited (the default: + # cover everything). Set it only for a pathological corpus -- a capped scan + # reports coverage='bounded' and discloses the scanned/total ratio, exactly + # as the old recency window did. + recall_scan_max_rows: int = 0 + # Optional hard latency budget for the lcm_recall scan, in seconds; 0 = no + # early stop (the default). When set, a scan that overruns it stops between + # batches and degrades to coverage='bounded' rather than silently paying an + # unbounded cost. This is the ONLY thing that truncates a default scan. + recall_scan_budget_s: float = 0.0 # Per-arm RRF fusion weights for lcm_recall's 3-arm hybrid (fts/summary/chunk). # Down-weighting the weak FTS arm keeps naive equal-weight fusion from dragging # fused recall below its best (vector) arm — measured −21 R@5 on LongMemEval. diff --git a/retrieval_core.py b/retrieval_core.py index e3d109916..90f9cb920 100644 --- a/retrieval_core.py +++ b/retrieval_core.py @@ -232,6 +232,9 @@ def run_knn( source: str | None, vector_store_cls: Any, scan_rows: int | None = None, + full_scan: bool = False, + scan_max_rows: int = 0, + scan_budget_s: float = 0.0, ) -> Any: """Run the vector KNN query inside the operation's absolute deadline. @@ -240,6 +243,9 @@ def run_knn( overrides the candidate-scan bound when set (``None`` keeps the configured ``embedding_bounded_scan_rows`` — the lcm_grep contract is unchanged); a cross-conversation caller passes a larger bound so "all time" is real. + ``full_scan`` turns that bound into a per-batch size and covers the whole + corpus (the lcm_recall contract), optionally capped by ``scan_max_rows`` / + ``scan_budget_s`` — both 0 (no early stop) by default. """ if time.monotonic() >= deadline: raise TimeoutError("semantic vector search deadline exhausted") @@ -257,6 +263,9 @@ def run_knn( until=until, conversation_ids=conversation_ids, source=source, + full_scan=full_scan, + scan_max_rows=scan_max_rows, + scan_budget_s=scan_budget_s, ), ) @@ -274,14 +283,17 @@ def run_chunk_knn( source: str | None, vector_store_cls: Any, scan_rows: int | None = None, + full_scan: bool = False, + scan_max_rows: int = 0, + scan_budget_s: float = 0.0, ) -> Any: """Run the chunk-corpus KNN query inside the operation's absolute deadline. Mirrors ``run_knn`` for the second (chunk) corpus: the same injected - ``vector_store_cls`` binding, ``scan_rows`` candidate-bound override, and - progress-handler deadline guard, calling ``knn_chunks`` instead of ``knn``. - Returns the store's coverage contract (full|bounded|none) so the caller - degrades identically to the summary arm. + ``vector_store_cls`` binding, ``scan_rows`` candidate-bound override, the + same ``full_scan`` batching, and progress-handler deadline guard, calling + ``knn_chunks`` instead of ``knn``. Returns the store's coverage contract + (full|bounded|none) so the caller degrades identically to the summary arm. """ if time.monotonic() >= deadline: raise TimeoutError("chunk vector search deadline exhausted") @@ -299,6 +311,9 @@ def run_chunk_knn( until=until, conversation_ids=conversation_ids, source=source, + full_scan=full_scan, + scan_max_rows=scan_max_rows, + scan_budget_s=scan_budget_s, ), ) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index 4f3369269..34a411730 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -994,9 +994,10 @@ def test_recall_uses_recall_timeout_budget(recall_engine, monkeypatch): def test_bounded_chunk_coverage_surfaces_as_degraded(recall_engine, monkeypatch): - """SCAN-1: a recency-bounded chunk arm reports a degraded_reasons entry naming - the arm + scanned/total, instead of silently truncating.""" - recall_engine._config.recall_scan_rows = 1 + """SCAN-1: a chunk arm capped by recall_scan_max_rows reports a + degraded_reasons entry naming the arm + scanned/total, instead of silently + truncating.""" + recall_engine._config.recall_scan_max_rows = 1 ids = [] for i in range(3): sid = recall_engine._store.append( @@ -1016,6 +1017,62 @@ def test_bounded_chunk_coverage_surfaces_as_degraded(recall_engine, monkeypatch) assert "of 3 vectors" in payload["degraded_reason"] +def test_recall_scan_batches_the_whole_corpus_and_reaches_the_oldest( + recall_engine, monkeypatch +): + """F31 §2: recall_scan_rows is a BATCH SIZE, not a recency window. + + The gold vector is the OLDEST message in a corpus more than 2x the batch + size — exactly the shape that collapsed all-gold recall to 0.000 when the + scan only scored the most-recent rows. + """ + recall_engine._config.recall_scan_rows = 2 # batch size; corpus is 7 vectors + seeded = [] + # The gold message is appended FIRST, so every later filler is more recent: + # under the old recency window it aged straight out of the scan. + gold = recall_engine._store.append( + CURRENT, {"role": "user", "content": "kanban dashboard sprint gold"} + ) + seeded.append((gold, [1.0, 0.0])) + for i in range(6): + sid = recall_engine._store.append( + CURRENT, {"role": "user", "content": f"unrelated filler note {i}"} + ) + seeded.append((sid, [0.0, 1.0])) + _seed_chunk_vectors( + recall_engine, + [(sid, 0, 0, 20, vector) for sid, vector in seeded], + ) + + payload = _recall(recall_engine, monkeypatch, include="verbatim", limit=1) + + assert payload["provenance"]["coverage"].get("chunk") == "full" + assert [hit["store_id"] for hit in payload["hits"]] == [gold] + + +def test_recall_scan_reports_no_degraded_reason_by_default(recall_engine, monkeypatch): + """Default config has no cap and no latency budget, so nothing truncates.""" + assert recall_engine._config.recall_scan_max_rows == 0 + assert recall_engine._config.recall_scan_budget_s == 0.0 + recall_engine._config.recall_scan_rows = 1 # one vector per batch + ids = [ + recall_engine._store.append( + CURRENT, {"role": "user", "content": f"kanban dashboard sprint chunk {i}"} + ) + for i in range(5) + ] + _seed_chunk_vectors( + recall_engine, + [(sid, 0, 0, 20, [1.0, 0.0]) for sid in ids], + ) + + payload = _recall(recall_engine, monkeypatch, include="verbatim", limit=10) + + assert payload["provenance"]["coverage"].get("chunk") == "full" + assert "degraded_reason" not in payload + assert {hit["store_id"] for hit in payload["hits"]} == set(ids) + + def test_two_stage_full_approx_coverage_surfaces_as_approximate(recall_engine, monkeypatch): """FIX 2: a two-stage (binary prescreen) summary arm reaches the whole corpus but ranks approximately, so it reports coverage='full_approx' and discloses diff --git a/tests/test_vector_store.py b/tests/test_vector_store.py index eee5334bd..f24ec6d12 100644 --- a/tests/test_vector_store.py +++ b/tests/test_vector_store.py @@ -392,6 +392,122 @@ def unavailable(): +def _seed_scan_corpus(dag, store, size, *, gold_vector, filler_vector): + """Seed ``size`` summaries whose OLDEST carries ``gold_vector``.""" + store.register_profile("scan", "local", 3) + gold = _add_summary(dag, created_at=1.0) + _record_embedding(store, gold, "summary", "scan", gold_vector) + for index in range(1, size): + node = _add_summary(dag, created_at=1.0 + index) + _record_embedding(store, node, "summary", "scan", filler_vector) + return gold + + +def test_full_scan_batches_the_whole_corpus_and_reaches_the_oldest(tmp_path): + """F31 §2: with full_scan the bound is a BATCH SIZE, not a recency window. + + The gold vector is the OLDEST of a corpus more than 2x the batch size — + the exact shape that made 86% of a 185k-vector store invisible and drove + all-gold recall to 0.000 as content aged out. + """ + db_path = tmp_path / "full-scan.db" + dag = SummaryDAG(db_path) + store = VectorStore(db_path, bounded_scan_rows=3) # 3-row batches, 7 vectors + try: + gold = _seed_scan_corpus( + dag, store, 7, gold_vector=[1.0, 0.0, 0.0], filler_vector=[0.0, 1.0, 0.0] + ) + + windowed = store.knn([1.0, 0.0, 0.0], k=1, model="scan") + assert windowed.coverage == "bounded" + assert [row[0] for row in windowed] != [str(gold)] # aged out of the window + + result = store.knn([1.0, 0.0, 0.0], k=1, model="scan", full_scan=True) + + assert result.coverage == "full" + assert result.reason is None + assert [row[0] for row in result] == [str(gold)] + finally: + store.close() + dag.close() + + +def test_full_scan_without_numpy_also_reaches_the_oldest(tmp_path, monkeypatch): + """The pure-Python scoring path batches identically (no numpy install).""" + db_path = tmp_path / "full-scan-nonumpy.db" + dag = SummaryDAG(db_path) + store = VectorStore(db_path, bounded_scan_rows=2) + try: + gold = _seed_scan_corpus( + dag, store, 5, gold_vector=[1.0, 0.0, 0.0], filler_vector=[0.0, 1.0, 0.0] + ) + + def unavailable(): + raise ImportError("numpy not installed") + + monkeypatch.setattr(vector_store_module, "_load_numpy", unavailable) + result = store.knn([1.0, 0.0, 0.0], k=1, model="scan", full_scan=True) + + assert result.coverage == "full" + assert [row[0] for row in result] == [str(gold)] + finally: + store.close() + dag.close() + + +def test_full_scan_max_rows_caps_the_scan_and_discloses_it(tmp_path): + """scan_max_rows is the pathological-corpus escape hatch, and it discloses.""" + db_path = tmp_path / "full-scan-capped.db" + dag = SummaryDAG(db_path) + store = VectorStore(db_path, bounded_scan_rows=2) + try: + gold = _seed_scan_corpus( + dag, store, 6, gold_vector=[1.0, 0.0, 0.0], filler_vector=[0.0, 1.0, 0.0] + ) + + result = store.knn( + [1.0, 0.0, 0.0], k=1, model="scan", full_scan=True, scan_max_rows=2 + ) + + assert result.coverage == "bounded" + assert result.scanned == 2 + assert result.total == 6 + assert [row[0] for row in result] != [str(gold)] + finally: + store.close() + dag.close() + + +def test_full_scan_budget_stops_early_and_reports_bounded(tmp_path, monkeypatch): + """An exhausted latency budget is the only thing that truncates a default + scan, and it degrades to the same disclosed 'bounded' coverage.""" + db_path = tmp_path / "full-scan-budget.db" + dag = SummaryDAG(db_path) + store = VectorStore(db_path, bounded_scan_rows=2) + try: + gold = _seed_scan_corpus( + dag, store, 6, gold_vector=[1.0, 0.0, 0.0], filler_vector=[0.0, 1.0, 0.0] + ) + # A clock that advances a second per reading: the budget is spent after + # the first batch, so the scan stops with 4 of 6 vectors unscored. + ticks = iter(range(1_000)) + monkeypatch.setattr( + vector_store_module.time, "monotonic", lambda: float(next(ticks)) + ) + + result = store.knn( + [1.0, 0.0, 0.0], k=1, model="scan", full_scan=True, scan_budget_s=0.5 + ) + + assert result.coverage == "bounded" + assert result.scanned == 2 + assert result.total == 6 + assert [row[0] for row in result] != [str(gold)] + finally: + store.close() + dag.close() + + def test_bounded_scan_keeps_null_latest_at_legacy_rows_by_created_at( tmp_path, monkeypatch ): diff --git a/tools.py b/tools.py index a7e80bb7a..b8289e86f 100644 --- a/tools.py +++ b/tools.py @@ -3701,6 +3701,23 @@ def _lcm_recall_fts_arm( return hits, None +def _lcm_recall_scan_bounds(engine: "LCMEngine") -> dict[str, Any]: + """Full-corpus scan settings shared by both vector arms. + + lcm_recall promises "all conversations, all time", so both arms scan the + WHOLE corpus in ``recall_scan_rows``-sized batches. The optional cap and + latency budget default to 0 (no early stop); either one firing degrades the + arm to ``coverage='bounded'``, which the caller already discloses. + """ + return { + "full_scan": True, + "scan_max_rows": max(0, int(getattr(engine._config, "recall_scan_max_rows", 0))), + "scan_budget_s": max( + 0.0, float(getattr(engine._config, "recall_scan_budget_s", 0.0)) + ), + } + + def _lcm_recall_summary_arm( engine: "LCMEngine", *, @@ -3723,6 +3740,7 @@ def _lcm_recall_summary_arm( source=None, vector_store_cls=VectorStore, scan_rows=max(1, int(getattr(engine._config, "recall_scan_rows", 25_000))), + **_lcm_recall_scan_bounds(engine), ), remaining_s=deadline - time.monotonic(), name="lcm-recall-summary-knn", @@ -3779,6 +3797,7 @@ def _lcm_recall_chunk_arm( source=None, vector_store_cls=VectorStore, scan_rows=max(1, int(getattr(engine._config, "recall_scan_rows", 25_000))), + **_lcm_recall_scan_bounds(engine), ), remaining_s=deadline - time.monotonic(), name="lcm-recall-chunk-knn", diff --git a/vector_store.py b/vector_store.py index e5c5c9991..62e18f92a 100644 --- a/vector_store.py +++ b/vector_store.py @@ -13,6 +13,7 @@ import sqlite3 import struct import threading +import time import uuid from collections import OrderedDict from contextlib import contextmanager @@ -71,6 +72,10 @@ # identity hash), so their vectors never collide. _SUPPORTED_TASKS = frozenset({_DEFAULT_TASK, _CHUNK_TASK}) +# Candidate-enumeration sentinel: SQLite reads a negative LIMIT as "no limit", +# so a full-corpus scan enumerates every live candidate for the identity. +_SCAN_ALL_ROWS = -1 + # SQLite caps host parameters per statement (SQLITE_MAX_VARIABLE_NUMBER); a # single ``WHERE id IN (?, ?, ...)`` over tens of thousands of ids overflows it # (observed failure at ~33k ids). Candidate id resolution loads ids into a temp @@ -1322,8 +1327,11 @@ def _bounded_candidate_ids( what ``coverage='bounded'`` signals. The source-lineage filter is a recursive descendant walk, so it is applied to the already-bounded id set afterward (and fails closed when provenance is unverifiable). + + ``limit=_SCAN_ALL_ROWS`` enumerates every live candidate (the batched + full-corpus scan bounds memory per BATCH, not per corpus). """ - if limit <= 0: + if limit == 0: return [] if conversation_ids is not None and not list(conversation_ids): return [] @@ -1443,22 +1451,78 @@ def _load_vectors_for_ids( return rowids, out_ids, kinds, vectors @staticmethod + def _rank_key(row: tuple[int, str, float, str]) -> tuple[float, int, str]: + """Sort key for one ``(rowid, embedded_id, score, kind)`` candidate.""" + return (-float(row[2]), -int(row[0]), str(row[1])) + + @classmethod def _ranked( + cls, rowids: Sequence[int], embedded_ids: Sequence[str], kinds: Sequence[str], scores: Sequence[float], limit: int, ) -> list[tuple[str, float, str]]: - ranked = sorted( - zip(rowids, embedded_ids, scores, kinds), - key=lambda row: (-float(row[2]), -int(row[0]), str(row[1])), - ) + ranked = sorted(zip(rowids, embedded_ids, scores, kinds), key=cls._rank_key) return [ (str(embedded_id), float(score), str(kind)) for _, embedded_id, score, kind in ranked[:limit] ] + def _scan_ranked( + self, + *, + candidate_ids: Sequence[str], + batch_rows: int, + budget_s: float, + limit: int, + score_batch: Any, + ) -> tuple[list[tuple[str, float, str]], int, bool]: + """Score every candidate in ``batch_rows`` chunks, keeping a running top-k. + + ``batch_rows`` bounds PEAK MEMORY (one batch of vectors resident at a + time), not the corpus reached: the whole candidate set is scored, so a + memory older than the batch size stays retrievable. That is the fix for + the 25k recency window that made 86% of a 185k-vector corpus invisible + to semantic recall (FINDING-F31 §2). + + ``budget_s`` (0 = no early stop, the default) is the only thing that can + cut the scan short; when it does, the caller degrades to + ``coverage='bounded'`` and the existing disclosure names the ratio. + Returns ``(ranked top-k, candidates scored, stopped early)``. + """ + best: list[tuple[int, str, float, str]] = [] + scanned = 0 + stopped_early = False + started = time.monotonic() + for start in range(0, len(candidate_ids), batch_rows): + batch = candidate_ids[start:start + batch_rows] + rowids, embedded_ids, kinds, scores = score_batch(batch) + scanned += len(batch) + best.extend( + (int(rowid), str(embedded_id), float(score), str(kind)) + for rowid, embedded_id, score, kind in zip( + rowids, embedded_ids, scores, kinds + ) + ) + if len(best) > limit: + best.sort(key=self._rank_key) + del best[limit:] + exhausted = start + batch_rows >= len(candidate_ids) + if not exhausted and budget_s > 0 and (time.monotonic() - started) >= budget_s: + stopped_early = True + break + best.sort(key=self._rank_key) + return ( + [ + (embedded_id, score, kind) + for _, embedded_id, score, kind in best[:limit] + ], + scanned, + stopped_early, + ) + def _source_allowed_ids(self, table: str, source: str) -> set[str]: """Root candidate ids whose source subtree reaches a message with ``source``. @@ -1852,6 +1916,9 @@ def knn( conversation_ids: Sequence[str] | None = None, source: str | None = None, provider: str | None = None, + full_scan: bool = False, + scan_max_rows: int = 0, + scan_budget_s: float = 0.0, ) -> KNNResult: k = int(k) if k <= 0: @@ -1913,9 +1980,11 @@ def knn( # approximate (recall@M) result, not exact like the exact-scan 'full'. return KNNResult(candidates, coverage="full_approx") - limit = max(0, self.bounded_scan_rows) - # Probe at most bound+1 through the indexed candidate query. This - # determines full-vs-bounded coverage without COUNT(*) scanning the + probe_limit, scan_limit = self._scan_limits( + full_scan=full_scan, scan_max_rows=scan_max_rows + ) + # Probe one past the scan limit through the indexed candidate query. + # This determines full-vs-bounded coverage without COUNT(*) scanning the # entire identity on every request. try: probed_ids = self._bounded_candidate_ids( @@ -1924,62 +1993,84 @@ def knn( until=until, conversation_ids=conversation_ids, source=source, - limit=limit + 1, + limit=probe_limit, ) except _UnverifiableProvenance: return KNNResult(coverage="none", reason="unverifiable_provenance") if not probed_ids: return KNNResult(coverage="none") - bounded_ids = probed_ids[:limit] + scan_ids = probed_ids if scan_limit is None else probed_ids[:scan_limit] candidate_coverage = ( "bounded" - if source is not None or len(probed_ids) > limit + if source is not None + or (scan_limit is not None and len(probed_ids) > scan_limit) else "full" ) if numpy is not None: - rowids, embedded_ids, kinds, matrix = self._numpy_rows( - numpy, - identity, - dim, - bounded_ids, - dtype, - ) query_array = numpy.asarray(query, dtype=numpy.float32) - scores = matrix @ query_array - coverage = candidate_coverage + + def score_batch(batch_ids: Sequence[str]) -> tuple[Any, Any, Any, Any]: + rowids, embedded_ids, kinds, matrix = self._numpy_rows( + numpy, + identity, + dim, + batch_ids, + dtype, + ) + return rowids, embedded_ids, kinds, matrix @ query_array else: - # Bound the candidate enumeration at the SQL layer: the column - # filters + ORDER BY source latest_at DESC + LIMIT run inside SQLite, so - # neither the result set nor host memory enumerates the whole - # corpus. Filters live in the WHERE clause (applied before the - # bound), so a filtered match inside the bounded window is not lost; - # only the source-lineage walk runs on the already-bounded set. - rowids, embedded_ids, kinds, vectors = self._load_vectors_for_ids( - identity, - dim, - bounded_ids, - dtype, - ) - scores = [ - sum(value * query_value for value, query_value in zip(vector, query)) - for vector in vectors - ] - coverage = candidate_coverage - - candidates = self._ranked( - rowids, - embedded_ids, - kinds, - scores, - k, + # The candidate enumeration itself runs at the SQL layer: the column + # filters + ORDER BY source latest_at DESC run inside SQLite, and + # only ONE batch of vectors is decoded into host memory at a time. + # Filters live in the WHERE clause, so a filtered match is never + # lost; only the source-lineage walk runs on the enumerated set. + def score_batch(batch_ids: Sequence[str]) -> tuple[Any, Any, Any, Any]: + rowids, embedded_ids, kinds, vectors = self._load_vectors_for_ids( + identity, + dim, + batch_ids, + dtype, + ) + return rowids, embedded_ids, kinds, [ + sum(value * query_value for value, query_value in zip(vector, query)) + for vector in vectors + ] + + candidates, scanned_rows, stopped_early = self._scan_ranked( + candidate_ids=scan_ids, + batch_rows=max(1, self.bounded_scan_rows), + budget_s=scan_budget_s, + limit=k, + score_batch=score_batch, ) + coverage = "bounded" if stopped_early else candidate_coverage scanned = total = None if coverage == "bounded": - scanned = len(bounded_ids) + scanned = scanned_rows total = self._count_embedded_vectors(identity, chunk=False) return KNNResult(candidates, coverage=coverage, scanned=scanned, total=total) + def _scan_limits( + self, *, full_scan: bool, scan_max_rows: int + ) -> tuple[int, int | None]: + """Resolve ``(candidate probe limit, hard scan cap)`` for one KNN call. + + Default (``full_scan=False``) keeps the historical recency window: + ``bounded_scan_rows`` candidates, probed one deeper so a larger corpus + reports ``coverage='bounded'``. ``full_scan=True`` (the lcm_recall + contract) enumerates the WHOLE corpus and uses ``bounded_scan_rows`` + only as the scan's batch size; ``scan_max_rows`` (0 = unlimited) is the + escape hatch for a pathological corpus and is disclosed as bounded + exactly like the recency window was. + """ + if not full_scan: + bound = max(0, self.bounded_scan_rows) + return bound + 1, bound + if scan_max_rows > 0: + return scan_max_rows + 1, scan_max_rows + return _SCAN_ALL_ROWS, None + # -- Chunk corpus ------------------------------------------------------ # # The chunk corpus mirrors the summary corpus exactly: identity-hashed @@ -2263,8 +2354,11 @@ def _bounded_chunk_candidate_ids( chunk maps DIRECTLY to one raw message, so the source filter is a plain indexed column match (no recursive lineage walk) and needs no fail-open/closed budget. + + ``limit=_SCAN_ALL_ROWS`` enumerates every live candidate (the batched + full-corpus scan bounds memory per BATCH, not per corpus). """ - if limit <= 0: + if limit == 0: return [] if conversation_ids is not None and not list(conversation_ids): return [] @@ -2381,13 +2475,16 @@ def knn_chunks( conversation_ids: Sequence[str] | None = None, source: str | None = None, provider: str | None = None, + full_scan: bool = False, + scan_max_rows: int = 0, + scan_budget_s: float = 0.0, ) -> KNNResult: """Bounded-candidate chunk KNN with the summary coverage contract. Coverage is full|bounded|none exactly as for summaries: ``none`` when the corpus/identity is unbackfilled or a requested filter is - unverifiable (missing message column), ``bounded`` when more candidates - exist than the scan bound, ``full`` otherwise. + unverifiable (missing message column), ``bounded`` when the scan was + cut short by a hard cap or latency budget, ``full`` otherwise. """ k = int(k) if k <= 0: @@ -2446,41 +2543,55 @@ def knn_chunks( # keeps only M=mult*k survivors -> approximate top-k, not exact. return KNNResult(candidates, coverage="full_approx") - limit = max(0, self.bounded_scan_rows) + probe_limit, scan_limit = self._scan_limits( + full_scan=full_scan, scan_max_rows=scan_max_rows + ) probed_ids = self._bounded_chunk_candidate_ids( identity, since=since, until=until, conversation_ids=conversation_ids, source=source, - limit=limit + 1, + limit=probe_limit, ) if not probed_ids: return KNNResult(coverage="none") - bounded_ids = probed_ids[:limit] - candidate_coverage = "bounded" if len(probed_ids) > limit else "full" + scan_ids = probed_ids if scan_limit is None else probed_ids[:scan_limit] + candidate_coverage = ( + "bounded" + if scan_limit is not None and len(probed_ids) > scan_limit + else "full" + ) if numpy is not None: - rowids, chunk_ids, kinds, matrix = self._numpy_chunk_rows( - numpy, identity, dim, bounded_ids, dtype - ) query_array = numpy.asarray(query, dtype=numpy.float32) - scores = matrix @ query_array - coverage = candidate_coverage + + def score_batch(batch_ids: Sequence[str]) -> tuple[Any, Any, Any, Any]: + rowids, chunk_ids, kinds, matrix = self._numpy_chunk_rows( + numpy, identity, dim, batch_ids, dtype + ) + return rowids, chunk_ids, kinds, matrix @ query_array else: - rowids, chunk_ids, kinds, vectors = self._load_chunk_vectors_for_ids( - identity, dim, bounded_ids, dtype - ) - scores = [ - sum(value * query_value for value, query_value in zip(vector, query)) - for vector in vectors - ] - coverage = candidate_coverage + def score_batch(batch_ids: Sequence[str]) -> tuple[Any, Any, Any, Any]: + rowids, chunk_ids, kinds, vectors = self._load_chunk_vectors_for_ids( + identity, dim, batch_ids, dtype + ) + return rowids, chunk_ids, kinds, [ + sum(value * query_value for value, query_value in zip(vector, query)) + for vector in vectors + ] - candidates = self._ranked(rowids, chunk_ids, kinds, scores, k) + candidates, scanned_rows, stopped_early = self._scan_ranked( + candidate_ids=scan_ids, + batch_rows=max(1, self.bounded_scan_rows), + budget_s=scan_budget_s, + limit=k, + score_batch=score_batch, + ) + coverage = "bounded" if stopped_early else candidate_coverage scanned = total = None if coverage == "bounded": - scanned = len(bounded_ids) + scanned = scanned_rows total = self._count_embedded_vectors(identity, chunk=True) return KNNResult(candidates, coverage=coverage, scanned=scanned, total=total) From c13f4a2d4264a752fb490e0dd9ddef083fed0c05 Mon Sep 17 00:00:00 2001 From: EVA Date: Wed, 29 Jul 2026 03:25:06 +0700 Subject: [PATCH 22/54] review finding 1: judge the LIKE fallback on the SANITIZED query, not the raw one requires_like_fallback() tested the RAW query, so a compound token still forced the full-table LIKE scan even though sanitization turns it into ordinary terms the index answers: requires_like_fallback("art-related") was True while sanitize_fts5_query("art-related") == "art related". Six of the fifty fixed Phase 1B questions carry a hyphen, so 12% of the rerun would have taken the same full-scan path this branch exists to remove. The predicate now asks what sanitization LOSES, not what the FTS5 query grammar cannot spell. Compounds ride the index. The genuine losses still route to LIKE and are tested against the RAW query, because sanitization is exactly what removes them: unicode61 does not segment CJK and drops emoji from the index, and a query that sanitizes to nothing has no term left to match. Four existing tests used a hyphen purely as their LIKE trigger; they now trigger on an emoji and keep the hyphen in the raw query, so the SQL-limit, conversational -vs-tool ranking, and risky-ASCII repetition collapse each still assert exactly what they did before. --- REVIEW-PACKET.md | 47 ++++++++++++++++++++++++++++++++++++++++ dag.py | 10 ++++----- search_query.py | 25 +++++++++++++++++++-- store.py | 10 ++++----- tests/test_lcm_core.py | 49 ++++++++++++++++++++++++++++++++++++++---- 5 files changed, 125 insertions(+), 16 deletions(-) create mode 100644 REVIEW-PACKET.md diff --git a/REVIEW-PACKET.md b/REVIEW-PACKET.md new file mode 100644 index 000000000..7c4031d7e --- /dev/null +++ b/REVIEW-PACKET.md @@ -0,0 +1,47 @@ +# REVIEW PACKET — adversarial correctness review of PR #169 (review ONLY, no code changes) + +You are the cross-model reviewer of Claude-authored, release-critical fixes. Your job: find real defects. +Do NOT propose gratuitous hardening, style changes, or extra tests — only defects that would produce wrong +behavior, wrong Phase 1B measurements, or regressions. Verdict + findings with file:line and severity. + +## What this PR is +Two fixes on `fix/phase1b-scan-and-query` (diff = `git diff e99f342..HEAD`, read-only git allowed): +1. `b2f228c` (#168): FTS5 query sanitization moved into the product — any non-alphanumeric char outside a + balanced phrase quote becomes a separator (matching unicode61 tokenizer splitting), applied at both FTS + entry points (store.py messages, dag.py summaries); LIKE fallback only for CJK/emoji/empty-after-sanitize. +2. `f960d9f` (#167): the 25k-most-recent vector scan window replaced by a batched FULL scan + (recall_scan_rows = batch size; running top-k across batches; new recall_scan_max_rows (0=unlimited) and + recall_scan_budget_s (0.0=no stop); degraded/degraded_reason only on actual truncation; lcm_grep path + byte-identical by default). + +## Context you should trust +Motivating measurements: FINDING-F31 in /Volumes/LEXAR/hermes-work/wt-ci-fix/bench/ — recall hit 0.000 at +185k vectors because golds aged out of the 25k window; raw questions returned 100% empty at scale via +FTS5-reject → LIKE full-scan → 8s timeout. Acceptance for the fixes is a zero-LLM instrument re-run +(Phase 1B), NOT this PR's tests — so your review should especially protect measurement correctness. + +## Attack surfaces (minimum) +1. **Sanitizer correctness**: characters FTS5 rejects vs what the transform handles; phrase-quote balancing + edge cases (unbalanced quote, quote-inside-word, empty phrase); can ANY input still reach fts5 MATCH as a + syntax error or as an unintended OPERATOR (NEAR/AND/OR/NOT semantics, column filters `col:`, `*` prefix)? + Does term-splitting match unicode61 exactly (unicode categories, diacritics, underscores, digits)? Is the + LIKE fallback condition right (CJK detection)? +2. **Batched top-k merge**: correctness across batch boundaries (ties, score ordering, duplicate rowids), + the final k composition vs the old single-window semantics on small corpora (must be byte-identical for + <= one batch); memory at 185k×384 (peak allocation per batch, no accidental full materialization); + numpy vs pure-Python path divergence; recall_scan_budget_s early-stop: is the partial result labelled + degraded EVERY time it truncates, and never otherwise? +3. **Behavior changes the author disclosed**: (a) `_search_like` now scores punctuation-only queries on raw + text — regression risk for existing callers? (b) the rewritten test + `test_bounded_chunk_coverage_surfaces_as_degraded` — does the new version still test the same contract? + (c) new config knob interactions (batch size 1, max_rows < batch, budget tiny). +4. **Caller sweep verification**: the author claims no product caller assumed the 25k cap — verify + independently (grep/trace knn/knn_chunks/recall_scan_rows consumers). +5. **Phase 1B instrument safety**: anything in these diffs that changes what `degraded_reason` strings say + (the instrument parses/records them), or that alters latency semantics in a way that would make the + Phase 1B latency curve incomparable to F31's (e.g., per-batch cache interactions the author flagged). + +## Output format +VERDICT: APPROVE / APPROVE-WITH-FIXES (list mandatory) / REJECT (why), then numbered findings: +severity CRITICAL/HIGH/MED/LOW · file:line · the defect · a minimal repro or reasoning. Max 15 findings, +real ones only. End with: the 3 riskiest lines of the diff, quoted. diff --git a/dag.py b/dag.py index b20ad929a..202c45d73 100644 --- a/dag.py +++ b/dag.py @@ -564,11 +564,11 @@ def search(self, query: str, session_id: str | None = None, safe_query = sanitize_fts5_query(query) terms = extract_search_terms(safe_query) phrases = extract_quoted_phrases(safe_query) - # LIKE is the fallback for text FTS cannot express (CJK/emoji/compound - # tokens) and for a query with no term left after sanitization. A raw - # natural-language question is NOT one of those: it sanitizes to a term - # form the index answers, so it stays on the FTS path (F31 §3). - if requires_like_fallback(query) or not safe_query: + # LIKE is the fallback for text sanitization LOSES (CJK/emoji) and for a + # query with no term left after it. A raw natural-language question is + # NOT one of those: it sanitizes to a term form the index answers, so it + # stays on the FTS path (F31 §3). + if requires_like_fallback(query, safe_query): return self._search_like(query, session_id=session_id, limit=limit, sort=sort, source=source) order_by = _build_search_order_by(sort, "COALESCE(n.latest_at, n.created_at)") diff --git a/search_query.py b/search_query.py index 4eaedaa6c..6af9996ec 100644 --- a/search_query.py +++ b/search_query.py @@ -100,8 +100,29 @@ def contains_risky_fts_ascii(text: str) -> bool: return bool(_RISKY_FTS_TOKEN_RE.search(text_without_phrases)) -def requires_like_fallback(query: str) -> bool: - return contains_cjk(query) or contains_emoji(query) or contains_risky_fts_ascii(query) +def requires_like_fallback(query: str, sanitized: str | None = None) -> bool: + """Whether ``query`` must be answered by the LIKE scan instead of the index. + + The test is what SANITIZATION LOSES, not what the FTS5 query grammar cannot + spell. A compound token (``art-related``, ``api:v2``, ``a/b``) sanitizes to + ordinary terms the index answers perfectly well, so routing it to the + full-table LIKE scan just re-imports the scaling ceiling this branch is + fixing — 6 of the 50 fixed Phase 1B questions carry a hyphen. The risky-ASCII + check therefore runs against the SANITIZED form, which is what actually + reaches ``MATCH``. + + Genuine losses stay on LIKE: unicode61 does not segment CJK, it drops emoji + from the index entirely, and a query that sanitizes to nothing has no terms + left to match. Those two character classes are tested against the RAW query + because sanitization is exactly what removes them. + """ + raw = query or "" + safe = sanitize_fts5_query(raw) if sanitized is None else (sanitized or "") + if not safe.strip(): + return True + if contains_cjk(raw) or contains_emoji(raw): + return True + return contains_risky_fts_ascii(safe) def _token_variants(token: str) -> List[str]: diff --git a/store.py b/store.py index 1d847d651..565c97212 100644 --- a/store.py +++ b/store.py @@ -1112,11 +1112,11 @@ def search(self, query: str, session_id: str | None = None, safe_query = sanitize_fts5_query(query) terms = extract_search_terms(safe_query) phrases = extract_quoted_phrases(safe_query) - # LIKE is the fallback for text FTS cannot express (CJK/emoji/compound - # tokens) and for a query with no term left after sanitization. A raw - # natural-language question is NOT one of those: it sanitizes to a term - # form the index answers, so it stays on the FTS path (F31 §3). - if requires_like_fallback(query) or not safe_query: + # LIKE is the fallback for text sanitization LOSES (CJK/emoji) and for a + # query with no term left after it. A raw natural-language question is + # NOT one of those: it sanitizes to a term form the index answers, so it + # stays on the FTS path (F31 §3). + if requires_like_fallback(query, safe_query): return self._search_like( query, session_id=session_id, diff --git a/tests/test_lcm_core.py b/tests/test_lcm_core.py index 6cb3e06e8..cda0b0809 100644 --- a/tests/test_lcm_core.py +++ b/tests/test_lcm_core.py @@ -1875,6 +1875,30 @@ def test_search_keeps_punctuated_question_on_the_fts_path(self, store): assert len(results) == 1 assert results[0]["content"] == "budget revenue q3 totals recorded" + def test_search_keeps_hyphenated_compound_on_the_fts_path(self, store): + """Review finding 1: a compound token sanitizes to ordinary terms, so it + must NOT be routed to the full-table LIKE scan (6 of the 50 fixed Phase + 1B questions carry a hyphen).""" + from hermes_lcm.search_query import requires_like_fallback + + assert requires_like_fallback("art-related") is False + store.append("sess1", {"role": "user", "content": "art related notes from tuesday"}) + store._search_like = lambda *args, **kwargs: pytest.fail( + "hyphenated compound fell back to the LIKE full-scan" + ) + + results = store.search("art-related", session_id="sess1") + + assert len(results) == 1 + assert results[0]["content"] == "art related notes from tuesday" + + def test_search_still_falls_back_to_like_for_cjk_and_emoji(self, store): + """The genuine losses stay on LIKE: sanitization cannot preserve them.""" + from hermes_lcm.search_query import requires_like_fallback + + assert requires_like_fallback("東京") is True + assert requires_like_fallback("launch \U0001F680") is True + def test_search_falls_back_to_like_when_query_sanitizes_empty(self, store): store.append("sess1", {"role": "user", "content": "what??? really"}) calls: list[str] = [] @@ -2184,7 +2208,11 @@ def test_search_like_fallback_applies_sql_limit(self, store): traced: list[str] = [] store._conn.set_trace_callback(traced.append) try: - results = store.search("plugin-only", session_id="sess1", limit=2, sort="relevance") + # The emoji is what routes to LIKE: a bare compound now sanitizes to + # terms the index answers (review finding 1). + results = store.search( + "plugin-only \U0001F680", session_id="sess1", limit=2, sort="relevance" + ) finally: store._conn.set_trace_callback(None) @@ -2230,7 +2258,11 @@ def test_search_prefers_conversational_hits_over_tool_output_noise(self, store): "content": '{"query":"hermes-lcm","matches":["hermes-lcm","hermes-lcm"]}', }, ) - fallback_results = store.search("hermes-lcm", session_id="sess1", limit=2, sort="relevance") + # The emoji is what routes to LIKE: a bare compound now sanitizes to + # terms the index answers (review finding 1). + fallback_results = store.search( + "hermes-lcm \U0001F680", session_id="sess1", limit=2, sort="relevance" + ) assert fallback_results[0]["store_id"] == fallback_user_id assert fallback_results[1]["store_id"] == fallback_tool_id @@ -3856,7 +3888,11 @@ def test_search_like_fallback_applies_sql_limit(self, dag): traced: list[str] = [] dag._conn.set_trace_callback(traced.append) try: - results = dag.search("plugin-only", session_id="s1", limit=2, sort="relevance") + # The emoji is what routes to LIKE: a bare compound now sanitizes to + # terms the index answers (review finding 1). + results = dag.search( + "plugin-only \U0001F680", session_id="s1", limit=2, sort="relevance" + ) finally: dag._conn.set_trace_callback(None) @@ -3979,7 +4015,12 @@ def test_search_relevance_prefers_direct_summary_over_risky_ascii_repetition_spa latest_at=1_700_000_000, )) - results = dag.search("plugin-only", session_id="s1", limit=2, sort="relevance") + # The emoji is what routes to LIKE: a bare compound now sanitizes to + # terms the index answers (review finding 1). The raw query keeps its + # hyphen, so the risky-ASCII repetition collapse under test is unchanged. + results = dag.search( + "plugin-only \U0001F680", session_id="s1", limit=2, sort="relevance" + ) assert results[0].node_id == direct assert results[1].node_id == spammy From 902fc3753946789fb0d047c9e8295aacade34571 Mon Sep 17 00:00:00 2001 From: EVA Date: Wed, 29 Jul 2026 03:27:18 +0700 Subject: [PATCH 23/54] review finding 2: stream multi-batch scans past the matrix LRU MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 4-entry matrix cache and a sequential batch sweep are actively hostile to each other. The 185,175-vector corpus needs eight 25k batches; after a sweep the cache holds batches 4-7, and the next sweep starts at batch 0 and evicts each entry before reaching it — zero hits on every measured repetition, pure reload cost. Worse, it RETAINED four float32 matrices (153.6MB on the chunk arm alone), which is exactly the peak-memory bound the batching was introduced to provide. A sweep that needs more than one batch now loads each batch straight through (_load_matrix / _load_chunk_matrix, no cache interaction), so peak memory really is one batch. A single-batch scan — every lcm_grep call, and any corpus under the batch size — takes the cached path unchanged, so the pooled-store warm path keeps its F2-matrix-cache behavior byte for byte. Cache semantics for the Phase 1B curve, so F34 can read it: rungs whose corpus fits one batch are WARM across repetitions; any rung above the batch size is DETERMINISTICALLY COLD. At the default 25k batch that is the 19,829-session rung (185k vectors) cold and the rest warm. --- tests/test_vector_store.py | 45 ++++++++++++++++- vector_store.py | 101 ++++++++++++++++++++++++++++--------- 2 files changed, 119 insertions(+), 27 deletions(-) diff --git a/tests/test_vector_store.py b/tests/test_vector_store.py index f24ec6d12..f119a5911 100644 --- a/tests/test_vector_store.py +++ b/tests/test_vector_store.py @@ -455,6 +455,47 @@ def unavailable(): dag.close() +def test_multi_batch_scan_does_not_populate_or_thrash_the_matrix_cache(tmp_path): + """Review finding 2: a sweep needing more batches than the 4-entry LRU + evicts each entry before the next sweep reaches it — zero hits, while + retaining 4 float32 matrices and breaking the one-batch memory bound. + Multi-batch sweeps therefore stream past the cache entirely.""" + db_path = tmp_path / "cache-thrash.db" + dag = SummaryDAG(db_path) + store = VectorStore(db_path, bounded_scan_rows=2) # 2-row batches, 12 vectors + try: + _seed_scan_corpus( + dag, store, 12, gold_vector=[1.0, 0.0, 0.0], filler_vector=[0.0, 1.0, 0.0] + ) + assert len(store._matrix_cache) == 0 + + for _ in range(3): # repeat: the old code re-loaded all 6 batches each time + store.knn([1.0, 0.0, 0.0], k=1, model="scan", full_scan=True) + + assert len(store._matrix_cache) == 0 + finally: + store.close() + dag.close() + + +def test_single_batch_scan_still_caches_its_matrix(tmp_path): + """The warm pooled-store path is unchanged when one batch covers the corpus.""" + db_path = tmp_path / "cache-single.db" + dag = SummaryDAG(db_path) + store = VectorStore(db_path, bounded_scan_rows=50) # one batch covers all 5 + try: + _seed_scan_corpus( + dag, store, 5, gold_vector=[1.0, 0.0, 0.0], filler_vector=[0.0, 1.0, 0.0] + ) + + store.knn([1.0, 0.0, 0.0], k=1, model="scan", full_scan=True) + + assert len(store._matrix_cache) == 1 + finally: + store.close() + dag.close() + + def test_full_scan_max_rows_caps_the_scan_and_discloses_it(tmp_path): """scan_max_rows is the pathological-corpus escape hatch, and it discloses.""" db_path = tmp_path / "full-scan-capped.db" @@ -1435,9 +1476,9 @@ def test_numpy_candidate_load_is_sql_bounded(tmp_path, monkeypatch): loaded: list[int] = [] original = store._numpy_rows - def counted(np, identity_hash, dim, ids, dtype="float32"): + def counted(np, identity_hash, dim, ids, dtype="float32", cache=True): loaded.append(len(ids)) - return original(np, identity_hash, dim, ids, dtype) + return original(np, identity_hash, dim, ids, dtype, cache=cache) monkeypatch.setattr(store, "_numpy_rows", counted) result = store.knn([1.0, 0.0], k=1, model="m") diff --git a/vector_store.py b/vector_store.py index 62e18f92a..c2a2e5570 100644 --- a/vector_store.py +++ b/vector_store.py @@ -1277,8 +1277,15 @@ def _numpy_rows( dim: int, embedded_ids: Sequence[str], dtype: str = _VECTOR_DTYPE, + cache: bool = True, ) -> tuple[list[int], list[str], list[str], Any]: - """Load only the SQL-bounded candidate set into a NumPy matrix.""" + """Load only the SQL-bounded candidate set into a NumPy matrix. + + ``cache=False`` streams the batch past the LRU entirely — see + ``_scan_ranked`` for why a multi-batch sweep must not cache. + """ + if not cache: + return self._load_matrix(numpy, identity_hash, dim, embedded_ids, dtype) with self._cache_lock: data_version = self._data_version(identity_hash) key = (identity_hash, data_version, tuple(str(value) for value in embedded_ids)) @@ -1286,20 +1293,31 @@ def _numpy_rows( if cached is not None: self._matrix_cache.move_to_end(key) # mark most-recently used return cached - rowids, loaded_ids, kinds, raw_vectors = self._load_vectors_for_ids( - identity_hash, dim, embedded_ids, dtype - ) - matrix = ( - numpy.asarray(raw_vectors, dtype=numpy.float32) - if raw_vectors - else numpy.empty((0, dim), dtype=numpy.float32) - ) - loaded = (rowids, loaded_ids, kinds, matrix) + loaded = self._load_matrix(numpy, identity_hash, dim, embedded_ids, dtype) self._matrix_cache[key] = loaded while len(self._matrix_cache) > self._MATRIX_CACHE_MAX_ENTRIES: self._matrix_cache.popitem(last=False) # evict oldest return loaded + def _load_matrix( + self, + numpy: Any, + identity_hash: str, + dim: int, + embedded_ids: Sequence[str], + dtype: str, + ) -> tuple[list[int], list[str], list[str], Any]: + """Decode one candidate set into a NumPy matrix (no cache interaction).""" + rowids, loaded_ids, kinds, raw_vectors = self._load_vectors_for_ids( + identity_hash, dim, embedded_ids, dtype + ) + matrix = ( + numpy.asarray(raw_vectors, dtype=numpy.float32) + if raw_vectors + else numpy.empty((0, dim), dtype=numpy.float32) + ) + return rowids, loaded_ids, kinds, matrix + def _bounded_candidate_ids( self, identity_hash: str, @@ -1491,14 +1509,24 @@ def _scan_ranked( cut the scan short; when it does, the caller degrades to ``coverage='bounded'`` and the existing disclosure names the ratio. Returns ``(ranked top-k, candidates scored, stopped early)``. + + A MULTI-BATCH sweep streams past the matrix LRU (``cache=False``). The + cache holds 4 entries, so a corpus needing more batches than that evicts + each one before the next sweep reaches it — every repetition pays a full + reload for zero hits, while still retaining 4 float32 matrices (153.6MB + on the 185k-vector chunk arm) and breaking the one-batch memory bound + this batching exists to provide. A single-batch scan — every lcm_grep + call and any corpus under the batch size — still caches exactly as + before, so the pooled-store warm path is unchanged. """ best: list[tuple[int, str, float, str]] = [] scanned = 0 stopped_early = False + cache_batches = len(candidate_ids) <= batch_rows started = time.monotonic() for start in range(0, len(candidate_ids), batch_rows): batch = candidate_ids[start:start + batch_rows] - rowids, embedded_ids, kinds, scores = score_batch(batch) + rowids, embedded_ids, kinds, scores = score_batch(batch, cache_batches) scanned += len(batch) best.extend( (int(rowid), str(embedded_id), float(score), str(kind)) @@ -2010,13 +2038,16 @@ def knn( if numpy is not None: query_array = numpy.asarray(query, dtype=numpy.float32) - def score_batch(batch_ids: Sequence[str]) -> tuple[Any, Any, Any, Any]: + def score_batch( + batch_ids: Sequence[str], cache: bool + ) -> tuple[Any, Any, Any, Any]: rowids, embedded_ids, kinds, matrix = self._numpy_rows( numpy, identity, dim, batch_ids, dtype, + cache=cache, ) return rowids, embedded_ids, kinds, matrix @ query_array else: @@ -2025,7 +2056,9 @@ def score_batch(batch_ids: Sequence[str]) -> tuple[Any, Any, Any, Any]: # only ONE batch of vectors is decoded into host memory at a time. # Filters live in the WHERE clause, so a filtered match is never # lost; only the source-lineage walk runs on the enumerated set. - def score_batch(batch_ids: Sequence[str]) -> tuple[Any, Any, Any, Any]: + def score_batch( + batch_ids: Sequence[str], _cache: bool + ) -> tuple[Any, Any, Any, Any]: rowids, embedded_ids, kinds, vectors = self._load_vectors_for_ids( identity, dim, @@ -2443,7 +2476,10 @@ def _numpy_chunk_rows( dim: int, chunk_ids: Sequence[str], dtype: str = _VECTOR_DTYPE, + cache: bool = True, ) -> tuple[list[int], list[str], list[str], Any]: + if not cache: + return self._load_chunk_matrix(numpy, identity_hash, dim, chunk_ids, dtype) with self._cache_lock: data_version = self._data_version(identity_hash) key = (identity_hash, data_version, tuple(str(value) for value in chunk_ids)) @@ -2451,20 +2487,31 @@ def _numpy_chunk_rows( if cached is not None: self._chunk_matrix_cache.move_to_end(key) # mark most-recently used return cached - rowids, loaded_ids, kinds, raw_vectors = self._load_chunk_vectors_for_ids( - identity_hash, dim, chunk_ids, dtype - ) - matrix = ( - numpy.asarray(raw_vectors, dtype=numpy.float32) - if raw_vectors - else numpy.empty((0, dim), dtype=numpy.float32) - ) - loaded = (rowids, loaded_ids, kinds, matrix) + loaded = self._load_chunk_matrix(numpy, identity_hash, dim, chunk_ids, dtype) self._chunk_matrix_cache[key] = loaded while len(self._chunk_matrix_cache) > self._MATRIX_CACHE_MAX_ENTRIES: self._chunk_matrix_cache.popitem(last=False) # evict oldest return loaded + def _load_chunk_matrix( + self, + numpy: Any, + identity_hash: str, + dim: int, + chunk_ids: Sequence[str], + dtype: str, + ) -> tuple[list[int], list[str], list[str], Any]: + """Decode one chunk candidate set into a NumPy matrix (no cache).""" + rowids, loaded_ids, kinds, raw_vectors = self._load_chunk_vectors_for_ids( + identity_hash, dim, chunk_ids, dtype + ) + matrix = ( + numpy.asarray(raw_vectors, dtype=numpy.float32) + if raw_vectors + else numpy.empty((0, dim), dtype=numpy.float32) + ) + return rowids, loaded_ids, kinds, matrix + def knn_chunks( self, query_vec: Sequence[float], @@ -2566,13 +2613,17 @@ def knn_chunks( if numpy is not None: query_array = numpy.asarray(query, dtype=numpy.float32) - def score_batch(batch_ids: Sequence[str]) -> tuple[Any, Any, Any, Any]: + def score_batch( + batch_ids: Sequence[str], cache: bool + ) -> tuple[Any, Any, Any, Any]: rowids, chunk_ids, kinds, matrix = self._numpy_chunk_rows( - numpy, identity, dim, batch_ids, dtype + numpy, identity, dim, batch_ids, dtype, cache=cache ) return rowids, chunk_ids, kinds, matrix @ query_array else: - def score_batch(batch_ids: Sequence[str]) -> tuple[Any, Any, Any, Any]: + def score_batch( + batch_ids: Sequence[str], _cache: bool + ) -> tuple[Any, Any, Any, Any]: rowids, chunk_ids, kinds, vectors = self._load_chunk_vectors_for_ids( identity, dim, batch_ids, dtype ) From c6fb1dab470af00cdb947bf33afce10671d95d37 Mon Sep 17 00:00:00 2001 From: EVA Date: Wed, 29 Jul 2026 03:28:36 +0700 Subject: [PATCH 24/54] review finding 3: route the scan limits through the binary-prescreen path too MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A fully-synced binary-prescreen identity returned coverage='full_approx' before _scan_limits() was ever consulted, so recall_scan_max_rows and recall_scan_budget_s were silently ignored on those profiles — on both the summary and chunk arms. An operator capping a pathological corpus got an uncapped scan and no disclosure that the cap did nothing. The two-stage path can honor neither knob: it is one matmul over the whole binary mirror, with no candidate window to cap and no batch boundary to stop at. So a requested bound now declines that path and takes the exact batched scan, which enforces the bound and reports the bounded coverage with scanned/total. Both knobs default to 0, so an unbounded recall keeps the two-stage path and its full_approx disclosure byte for byte. --- tests/test_prescreen_flip_blackout.py | 36 +++++++++++++++++++++++++++ vector_store.py | 34 +++++++++++++++++++++---- 2 files changed, 65 insertions(+), 5 deletions(-) diff --git a/tests/test_prescreen_flip_blackout.py b/tests/test_prescreen_flip_blackout.py index 2a0190111..462fa181f 100644 --- a/tests/test_prescreen_flip_blackout.py +++ b/tests/test_prescreen_flip_blackout.py @@ -137,3 +137,39 @@ def test_partial_binary_corpus_falls_back_to_exact_scan(tmp_path): assert result[0][0] == str(node_x) store.close() dag.close() + + +def test_scan_bounds_route_a_synced_binary_identity_to_the_exact_scan(tmp_path): + """Review finding 3: a fully-synced binary identity returned ``full_approx`` + BEFORE the scan limits were consulted, so recall_scan_max_rows and + recall_scan_budget_s were silently ignored on those profiles. A requested + hard bound must reach every scan path and report bounded coverage.""" + db_path = tmp_path / "bounded-prescreen.db" + dag = SummaryDAG(db_path) + config = LCMConfig(embedding_binary_prescreen=True) + store = VectorStore(db_path, config=config) + store.register_profile(MODEL, PROVIDER, DIM) + oldest = _add_summary(dag, created_at=1.0) + _record(store, oldest, [1.0, 0.0, 0.0]) + for index in range(1, 4): + _record(store, _add_summary(dag, created_at=1.0 + index), [0.0, 1.0, 0.0]) + assert store._binary_fully_synced(store._current_profile()["identity_hash"], chunk=False) + + # Unbounded: the two-stage path still runs and discloses its approximation. + unbounded = store.knn([1.0, 0.0, 0.0], k=1, model=MODEL, full_scan=True) + assert unbounded.coverage == "full_approx" + + capped = store.knn( + [1.0, 0.0, 0.0], k=1, model=MODEL, full_scan=True, scan_max_rows=1 + ) + assert capped.coverage == "bounded" + assert capped.scanned == 1 + assert capped.total == 4 + assert [row[0] for row in capped] != [str(oldest)] # oldest is outside the cap + + budgeted = store.knn( + [1.0, 0.0, 0.0], k=1, model=MODEL, full_scan=True, scan_budget_s=0.0001 + ) + assert budgeted.coverage in {"full", "bounded"} # exact path, never full_approx + store.close() + dag.close() diff --git a/vector_store.py b/vector_store.py index c2a2e5570..49ad504bf 100644 --- a/vector_store.py +++ b/vector_store.py @@ -1983,9 +1983,14 @@ def knn( # COMPLETE mirror of its vectors. A partial binary corpus (FIX 1) stays on # the exact scan below so no vector is silently INNER-JOINed away. The # source filter stays on the exact bounded path (its lineage walk is - # budget-bounded and does not scale to a full-corpus scan). - if numpy is not None and source is None and self._binary_fully_synced( - identity, chunk=False + # budget-bounded and does not scale to a full-corpus scan). A requested + # hard bound also routes to the exact path, which is the only one that + # can enforce and disclose it. + if ( + numpy is not None + and source is None + and not self._scan_bounds_requested(scan_max_rows, scan_budget_s) + and self._binary_fully_synced(identity, chunk=False) ): unfiltered = since is None and until is None and conversation_ids is None binary_ids, binary_matrix = self._cached_binary_matrix( @@ -2104,6 +2109,19 @@ def _scan_limits( return scan_max_rows + 1, scan_max_rows return _SCAN_ALL_ROWS, None + @staticmethod + def _scan_bounds_requested(scan_max_rows: int, scan_budget_s: float) -> bool: + """Whether the caller asked for a hard bound on this scan. + + The two-stage binary-prescreen path can honor neither: it is one matmul + over the whole binary mirror, with no candidate window to cap and no + batch boundary to stop at. When a bound is requested we therefore route + to the exact batched scan, which enforces it and reports the bounded + coverage. Both knobs default to 0, so an unbounded recall keeps the + two-stage path and its ``full_approx`` disclosure unchanged. + """ + return scan_max_rows > 0 or scan_budget_s > 0 + # -- Chunk corpus ------------------------------------------------------ # # The chunk corpus mirrors the summary corpus exactly: identity-hashed @@ -2565,8 +2583,14 @@ def knn_chunks( # is a COMPLETE mirror of its vectors (FIX 1 — a partial binary corpus # would be silently truncated by the INNER JOIN). Chunk filters are plain # indexed message columns (timestamp/session_id/source), so they all apply - # in the binary load over the whole (filtered) corpus. - if numpy is not None and self._binary_fully_synced(identity, chunk=True): + # in the binary load over the whole (filtered) corpus. A requested hard + # bound routes to the exact path, which is the only one that can enforce + # and disclose it. + if ( + numpy is not None + and not self._scan_bounds_requested(scan_max_rows, scan_budget_s) + and self._binary_fully_synced(identity, chunk=True) + ): unfiltered = ( since is None and until is None and conversation_ids is None and source is None From 4de8727ff1364af2512fc5d3ba095591bcf62ff1 Mon Sep 17 00:00:00 2001 From: EVA Date: Wed, 29 Jul 2026 03:30:13 +0700 Subject: [PATCH 25/54] review finding 4: keep emoji on the LIKE path it was routed to MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Sharing the FTS term form with the LIKE fallback deleted the very characters that fallback exists to find. `launch 🚀` routes to LIKE precisely BECAUSE unicode61 does not index the emoji — and then sanitized to `launch`, searching only %launch%. A row containing just 🚀 stopped matching. Regression against the base sanitizer, on both message and summary search. The LIKE path now has its own weaker sanitizer (sanitize_like_query), which strips only genuine FTS5 query-syntax operators — the punctuation a user typed FOR the index — and preserves every other character, because a character the index cannot spell is still a character a substring match can find. That is the pre-Phase-1B behavior, restored for the one path that wants it, while MATCH keeps the strict term form. Both walk the same quote-preserving scanner. This also retires the `or query` empty-sanitize workaround: sanitize_like_query never empties a punctuation-only query in the first place. --- dag.py | 7 ++++--- search_query.py | 42 +++++++++++++++++++++++++++++++++++------- store.py | 7 ++++--- tests/test_lcm_core.py | 31 +++++++++++++++++++++++++++++++ 4 files changed, 74 insertions(+), 13 deletions(-) diff --git a/dag.py b/dag.py index 202c45d73..f4abe0509 100644 --- a/dag.py +++ b/dag.py @@ -46,6 +46,7 @@ normalize_search_sort, requires_like_fallback, sanitize_fts5_query, + sanitize_like_query, should_apply_directness_rank_adjustment, ) from .store import _normalize_source_value, _UNKNOWN_SOURCE, _legacy_blank_source_clause @@ -642,9 +643,9 @@ def search(self, query: str, session_id: str | None = None, def _search_like(self, query: str, session_id: str | None = None, limit: int = 20, sort: str | None = None, source: str | None = None) -> List[SummaryNode]: - # A query that sanitizes away entirely (pure punctuation) still has a - # literal substring meaning here, so score it on the raw text. - safe_query = sanitize_fts5_query(query) or query + # LIKE keeps every character the index cannot spell (emoji, punctuation) + # because substring matching is the only way to find those rows. + safe_query = sanitize_like_query(query) terms = extract_search_terms(safe_query) phrases = extract_quoted_phrases(safe_query) if not terms: diff --git a/search_query.py b/search_query.py index 6af9996ec..7285690ad 100644 --- a/search_query.py +++ b/search_query.py @@ -3,7 +3,7 @@ from __future__ import annotations import re -from typing import List +from typing import Callable, List _CJK_RE = re.compile( r"[" @@ -26,6 +26,22 @@ _RISKY_FTS_TOKEN_RE = re.compile(r"[A-Za-z0-9][\-:/][A-Za-z0-9]") _SPLIT_PUNCT_RE = re.compile(r"[-:/]+") _STRIP_EDGE_PUNCT = "\"'()[]{}.,;" +# Characters that are special in FTS5 QUERY SYNTAX (as opposed to characters +# FTS5 simply cannot spell in a bareword). Only these have to go on the LIKE +# path, which has no query grammar of its own. +_FTS5_SPECIAL_CHARS = frozenset('"()*^-:{}.') + + +def _like_safe_char(char: str) -> str: + """Map one unquoted character to its LIKE-safe form. + + LIKE is a substring match, so the only thing it needs removed is the + operator punctuation a user typed FOR the index (quoted phrases, prefix + ``*``). Everything else is signal it can match on — emoji above all, which + the FTS term form must drop because unicode61 does not index it, and which + LIKE is therefore the ONLY way to find. + """ + return " " if char in _FTS5_SPECIAL_CHARS else char def _fts5_safe_char(char: str) -> str: @@ -44,12 +60,24 @@ def _fts5_safe_char(char: str) -> str: return char if (char.isalnum() or char.isspace()) else " " -def _sanitize_unquoted_fts5_fragment(text: str) -> str: - return "".join(_fts5_safe_char(char) for char in text) - - def sanitize_fts5_query(query: str) -> str: """Reduce a query to FTS5-safe terms, preserving balanced phrase quotes.""" + return _sanitize_query(query, _fts5_safe_char) + + +def sanitize_like_query(query: str) -> str: + """Strip FTS5 syntax operators, preserving every other character. + + The LIKE path's sanitization has to be WEAKER than the FTS one: a character + the index cannot spell is still a character LIKE can match. Sharing the FTS + term form here dropped emoji from the fallback that exists to find them + (``launch 🚀`` searched only ``%launch%``). + """ + return _sanitize_query(query, _like_safe_char) + + +def _sanitize_query(query: str, replace: Callable[[str], str]) -> str: + """Walk ``query`` outside balanced phrase quotes, mapping chars via ``replace``.""" if not query: return "" @@ -73,9 +101,9 @@ def sanitize_fts5_query(query: str) -> str: if in_quote: quote_buffer.append(char) continue - result.append(_fts5_safe_char(char)) + result.append(replace(char)) if in_quote and quote_buffer: - result.extend(_sanitize_unquoted_fts5_fragment("".join(quote_buffer))) + result.extend(replace(char) for char in "".join(quote_buffer)) return " ".join("".join(result).split()) diff --git a/store.py b/store.py index 565c97212..512a45991 100644 --- a/store.py +++ b/store.py @@ -44,6 +44,7 @@ normalize_search_sort, requires_like_fallback, sanitize_fts5_query, + sanitize_like_query, AGE_DECAY_RATE, should_apply_directness_rank_adjustment, ) @@ -1232,9 +1233,9 @@ def _search_like(self, query: str, session_id: str | None = None, role: str | None = None, time_from: float | None = None, time_to: float | None = None) -> List[Dict[str, Any]]: - # A query that sanitizes away entirely (pure punctuation) still has a - # literal substring meaning here, so score it on the raw text. - safe_query = sanitize_fts5_query(query) or query + # LIKE keeps every character the index cannot spell (emoji, punctuation) + # because substring matching is the only way to find those rows. + safe_query = sanitize_like_query(query) terms = extract_search_terms(safe_query) phrases = extract_quoted_phrases(safe_query) if not terms: diff --git a/tests/test_lcm_core.py b/tests/test_lcm_core.py index cda0b0809..f150fd208 100644 --- a/tests/test_lcm_core.py +++ b/tests/test_lcm_core.py @@ -1899,6 +1899,16 @@ def test_search_still_falls_back_to_like_for_cjk_and_emoji(self, store): assert requires_like_fallback("東京") is True assert requires_like_fallback("launch \U0001F680") is True + def test_like_fallback_keeps_the_emoji_it_was_routed_here_for(self, store): + """Review finding 4: `launch 🚀` routes to LIKE precisely BECAUSE the + index cannot hold the emoji — so the LIKE path must not then sanitize + the emoji away and search only `%launch%`.""" + emoji_only = store.append("sess1", {"role": "user", "content": "\U0001F680"}) + + results = store.search("launch \U0001F680", session_id="sess1") + + assert emoji_only in {row["store_id"] for row in results} + def test_search_falls_back_to_like_when_query_sanitizes_empty(self, store): store.append("sess1", {"role": "user", "content": "what??? really"}) calls: list[str] = [] @@ -7459,3 +7469,24 @@ def test_count_tokens_skips_lru_for_large_strings(monkeypatch): assert first == second assert tokens._count_tokens_cached.cache_info().currsize == 0 + + +class TestSummaryDagLikeSanitization: + """Review finding 4, summary side: the DAG LIKE fallback keeps its emoji.""" + + def test_like_fallback_keeps_the_emoji_it_was_routed_here_for(self, tmp_path): + dag = SummaryDAG(tmp_path / "emoji-dag.db") + try: + emoji_only = dag.add_node(SummaryNode( + session_id="s1", depth=0, summary="\U0001F680", + token_count=4, source_ids=[1], source_type="messages", + created_at=1_700_000_000, + earliest_at=1_700_000_000, + latest_at=1_700_000_000, + )) + + results = dag.search("launch \U0001F680", session_id="s1", limit=5) + + assert emoji_only in {node.node_id for node in results} + finally: + dag.close() From 530bd980ce4e26b5016dee1f9309d52cf079b018 Mon Sep 17 00:00:00 2001 From: EVA Date: Wed, 29 Jul 2026 03:33:04 +0700 Subject: [PATCH 26/54] review finding 5: neutralize bare boolean operators in the sanitized query MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Uppercase AND/OR/NOT/NEAR survived sanitization, so a raw question silently ACQUIRED boolean semantics it never asked for. `Portland, OR hotel` sanitized to `Portland OR hotel` and broadened into a disjunction; a leading `NOT ready` was an outright FTS syntax error that dumped the query onto the LIKE full-scan — the exact failure mode this branch exists to close. Raw and deliberate queries share one unmarked entry point, so the raw reading has to be the safe one: bare operators outside quoted phrases are lowercased, which turns them back into ordinary barewords (FTS5 operators are operators only in uppercase). An explicit "NEAR" phrase is already a literal and is untouched. The two hyphenated-operator tests asserted the OLD contract, where a stray OR broadened the query. They now assert the new one — the messy query resolves conjunctively with no syntax error and no full-table fallback — plus a positive check that the hyphenated compounds themselves reach the target through the index, which is what finding 1 bought. --- search_query.py | 31 ++++++++++++++++++++- tests/test_lcm_core.py | 63 ++++++++++++++++++++++++++++++++++++++---- 2 files changed, 88 insertions(+), 6 deletions(-) diff --git a/search_query.py b/search_query.py index 7285690ad..3b0feed11 100644 --- a/search_query.py +++ b/search_query.py @@ -60,9 +60,38 @@ def _fts5_safe_char(char: str) -> str: return char if (char.isalnum() or char.isspace()) else " " +def _lower_operator_tokens(text: str) -> str: + return " ".join( + token.lower() if token in _BOOLEAN_OPERATORS else token + for token in text.split(" ") + ) + + +def _neutralize_bare_operators(sanitized: str) -> str: + """Lowercase bare AND/OR/NOT/NEAR outside quoted phrases. + + FTS5 operators are only operators in UPPERCASE, so lowercasing turns them + back into ordinary barewords. Without this a raw question SILENTLY ACQUIRES + boolean semantics it never asked for: ``Portland, OR hotel`` sanitized to + ``Portland OR hotel`` and broadened to a disjunction, while a leading + ``NOT ready`` was a syntax error that dumped the query onto the LIKE + full-scan. Raw and deliberate queries share one unmarked entry point, so the + raw reading has to be the safe one. Quoted phrases are left alone: an + explicit ``"NEAR"`` is already a literal, not an operator. + """ + out: list[str] = [] + last = 0 + for match in _QUOTED_PHRASE_RE.finditer(sanitized): + out.append(_lower_operator_tokens(sanitized[last:match.start()])) + out.append(match.group(0)) + last = match.end() + out.append(_lower_operator_tokens(sanitized[last:])) + return "".join(out) + + def sanitize_fts5_query(query: str) -> str: """Reduce a query to FTS5-safe terms, preserving balanced phrase quotes.""" - return _sanitize_query(query, _fts5_safe_char) + return _neutralize_bare_operators(_sanitize_query(query, _fts5_safe_char)) def sanitize_like_query(query: str) -> str: diff --git a/tests/test_lcm_core.py b/tests/test_lcm_core.py index f150fd208..036ab9bf0 100644 --- a/tests/test_lcm_core.py +++ b/tests/test_lcm_core.py @@ -1899,6 +1899,27 @@ def test_search_still_falls_back_to_like_for_cjk_and_emoji(self, store): assert requires_like_fallback("東京") is True assert requires_like_fallback("launch \U0001F680") is True + def test_search_does_not_let_a_raw_query_acquire_or_semantics(self, store): + """Review finding 5: `Portland, OR hotel` must stay a conjunction of the + words the user typed, not broaden into a disjunction.""" + both = store.append("sess1", {"role": "user", "content": "portland or hotel notes"}) + store.append("sess1", {"role": "user", "content": "an unrelated hotel in denver"}) + + results = store.search("Portland, OR hotel", session_id="sess1") + + assert [row["store_id"] for row in results] == [both] + + def test_search_survives_a_leading_boolean_operator(self, store): + """`NOT ready` was an FTS syntax error that dumped the query on LIKE.""" + store.append("sess1", {"role": "user", "content": "not ready for launch"}) + store._search_like = lambda *args, **kwargs: pytest.fail( + "leading operator fell back to the LIKE full-scan" + ) + + results = store.search("NOT ready", session_id="sess1") + + assert len(results) == 1 + def test_like_fallback_keeps_the_emoji_it_was_routed_here_for(self, store): """Review finding 4: `launch 🚀` routes to LIKE precisely BECAUSE the index cannot hold the emoji — so the LIKE path must not then sanitize @@ -2207,9 +2228,23 @@ def test_search_hyphenated_operator_queries_fall_back_cleanly(self, store): query = "8416 OR vendored OR vendoring OR plugin-only OR external context-engine OR generic host support OR hermes-lcm stays external OR no vendoring" results = store.search(query, session_id="sess1", limit=5, sort="relevance") - assert len(results) == 1 - assert results[0]["store_id"] == target - assert results[0]["snippet"] + # Review findings 1 + 5: the compounds now ride the index and the bare + # OR is neutralized, so this is a conjunction of every word typed. No + # row satisfies all of them (`8416`, `vendored` appear nowhere). What + # "cleanly" now means is that it resolves without an FTS syntax error, + # without the full-table fallback, and without a stray OR silently + # broadening the query into the filler row. + assert results == [] + + # The compounds themselves still reach the target through the index. + hits = store.search( + "plugin-only context-engine hermes-lcm stays external", + session_id="sess1", + limit=5, + sort="relevance", + ) + assert [row["store_id"] for row in hits] == [target] + assert hits[0]["snippet"] def test_search_like_fallback_applies_sql_limit(self, store): for idx in range(80): @@ -3265,6 +3300,14 @@ def test_sanitize_fts5_query_leaves_clean_queries_unchanged(self): assert sanitize_fts5_query("docker deploy notes") == "docker deploy notes" assert sanitize_fts5_query("東京 memo") == "東京 memo" + def test_sanitize_fts5_query_neutralizes_bare_boolean_operators(self): + # Review finding 5: a raw question must never acquire operator semantics. + assert sanitize_fts5_query("Portland, OR hotel") == "Portland or hotel" + assert sanitize_fts5_query("NOT ready") == "not ready" + assert sanitize_fts5_query("cats AND dogs NEAR birds") == "cats and dogs near birds" + # An explicit phrase is already a literal, so it is left untouched. + assert sanitize_fts5_query('"NOT ready" today') == '"NOT ready" today' + def test_sanitize_fts5_query_empties_a_punctuation_only_query(self): assert sanitize_fts5_query("???") == "" assert sanitize_fts5_query("!!! ***") == "" @@ -3881,8 +3924,18 @@ def test_search_hyphenated_operator_queries_fall_back_cleanly(self, dag): query = "8416 OR vendored OR vendoring OR plugin-only OR external context-engine OR generic host support OR hermes-lcm stays external OR no vendoring" results = dag.search(query, session_id="s1", limit=5, sort="relevance") - assert len(results) == 1 - assert results[0].node_id == target + # Review findings 1 + 5: see the MessageStore twin. Conjunction of every + # word typed, so nothing matches; "cleanly" now means no syntax error, + # no full-table fallback, and no OR broadening into the filler node. + assert results == [] + + hits = dag.search( + "plugin-only context-engine hermes-lcm stays external", + session_id="s1", + limit=5, + sort="relevance", + ) + assert [node.node_id for node in hits] == [target] def test_search_like_fallback_applies_sql_limit(self, dag): for idx in range(80): From 5b5815402226b28091c05c9f5ab64014c4f531ed Mon Sep 17 00:00:00 2001 From: EVA Date: Wed, 29 Jul 2026 03:35:09 +0700 Subject: [PATCH 27/54] review finding 6: compose the query before splitting it into FTS terms MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit str.isalnum() is not unicode61's token boundary. A combining mark is not alphanumeric, so a decomposed "naïve" (nai + U+0308) sanitized to "nai ve" while unicode61 folds and indexes the word as "naive" — zero rows for a query that matched raw. That directly contradicted the tokenizer-parity invariant this sanitizer is justified by. sanitize_fts5_query now composes to NFC first, so a decomposed accent is the one alphanumeric character the tokenizer folds, and any mark that does NOT compose (a virama, stacked diacritics) is kept inside its token rather than treated as a separator. The LIKE path deliberately does not normalize: it is a literal substring match against stored bytes and must not re-spell the query. --- search_query.py | 22 +++++++++++++++++++--- tests/test_lcm_core.py | 26 ++++++++++++++++++++++++++ 2 files changed, 45 insertions(+), 3 deletions(-) diff --git a/search_query.py b/search_query.py index 3b0feed11..2551a5ab6 100644 --- a/search_query.py +++ b/search_query.py @@ -3,6 +3,7 @@ from __future__ import annotations import re +import unicodedata from typing import Callable, List _CJK_RE = re.compile( @@ -56,8 +57,16 @@ def _fts5_safe_char(char: str) -> str: the index in its term form; the default unicode61 tokenizer splits the INDEXED text on exactly the same boundary, so no term is lost by the substitution. + + ``str.isalnum()`` alone is NOT that boundary: a combining mark is not + alphanumeric, so a decomposed ``naïve`` (``nai`` + U+0308) split into + ``nai ve`` while unicode61 indexes the word as ``naive`` — zero rows where + the raw query matched. Marks therefore stay inside the token, and + ``sanitize_fts5_query`` composes the query first. """ - return char if (char.isalnum() or char.isspace()) else " " + if char.isalnum() or char.isspace(): + return char + return char if unicodedata.category(char).startswith("M") else " " def _lower_operator_tokens(text: str) -> str: @@ -90,8 +99,15 @@ def _neutralize_bare_operators(sanitized: str) -> str: def sanitize_fts5_query(query: str) -> str: - """Reduce a query to FTS5-safe terms, preserving balanced phrase quotes.""" - return _neutralize_bare_operators(_sanitize_query(query, _fts5_safe_char)) + """Reduce a query to FTS5-safe terms, preserving balanced phrase quotes. + + Composed (NFC) first so a decomposed accent is one alphanumeric character + rather than a base plus a combining mark, which is what unicode61 folds and + indexes. The LIKE path deliberately does NOT normalize: it is a literal + substring match against stored bytes, so it must not re-spell the query. + """ + composed = unicodedata.normalize("NFC", query or "") + return _neutralize_bare_operators(_sanitize_query(composed, _fts5_safe_char)) def sanitize_like_query(query: str) -> str: diff --git a/tests/test_lcm_core.py b/tests/test_lcm_core.py index 036ab9bf0..b8cb66c3f 100644 --- a/tests/test_lcm_core.py +++ b/tests/test_lcm_core.py @@ -8,6 +8,7 @@ import sys import threading import time +import unicodedata from pathlib import Path from types import ModuleType, SimpleNamespace @@ -1899,6 +1900,18 @@ def test_search_still_falls_back_to_like_for_cjk_and_emoji(self, store): assert requires_like_fallback("東京") is True assert requires_like_fallback("launch \U0001F680") is True + def test_search_matches_a_decomposed_accent_against_the_index(self, store): + """Review finding 6: unicode61 folds `naïve` to `naive`, so a decomposed + query must compose rather than split into `nai ve` and match nothing.""" + target = store.append("sess1", {"role": "user", "content": "a naïve approach"}) + store._search_like = lambda *args, **kwargs: pytest.fail( + "decomposed accent fell back to the LIKE full-scan" + ) + + results = store.search("nai\u0308ve", session_id="sess1") # NFD + + assert [row["store_id"] for row in results] == [target] + def test_search_does_not_let_a_raw_query_acquire_or_semantics(self, store): """Review finding 5: `Portland, OR hotel` must stay a conjunction of the words the user typed, not broaden into a disjunction.""" @@ -3300,6 +3313,19 @@ def test_sanitize_fts5_query_leaves_clean_queries_unchanged(self): assert sanitize_fts5_query("docker deploy notes") == "docker deploy notes" assert sanitize_fts5_query("東京 memo") == "東京 memo" + def test_sanitize_fts5_query_composes_decomposed_accents(self): + # Review finding 6: str.isalnum() is not unicode61's boundary. A + # decomposed accent split the token ("nai ve") while the index holds + # the folded "naive", so the sanitized query matched zero rows. + decomposed = "naïve" # NFD: base + combining diaeresis + assert len(decomposed) == 6 + sanitized = sanitize_fts5_query(decomposed) + assert sanitized == "naïve" # NFC, one token, no split + assert sanitized == unicodedata.normalize("NFC", decomposed) + # A combining mark that does not compose stays inside its token. + virama = "क्ष" + assert sanitize_fts5_query(virama) == virama + def test_sanitize_fts5_query_neutralizes_bare_boolean_operators(self): # Review finding 5: a raw question must never acquire operator semantics. assert sanitize_fts5_query("Portland, OR hotel") == "Portland or hotel" From e4577f19892791c3a22c6b9efed38f25bceb89d8 Mon Sep 17 00:00:00 2001 From: EVA Date: Wed, 29 Jul 2026 03:38:36 +0700 Subject: [PATCH 28/54] review finding 5 (follow-up): mark the deliberate-operator mode instead of guessing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Neutralizing bare operators broke the ONE in-repo caller that composes FTS5 syntax on purpose: benchmarking/longmemeval.build_fts_query joins its barewords with OR, and lowercasing them turned the harness's disjunction into a conjunction that also required the literal word "or" — the FTS arm's recall@10 fell from 1.0 to 0.0. That is the Phase 1B instrument itself, so leaving it broken would have silently gutted the rerun this branch exists to enable. The reviewer named the real defect: raw and deliberate queries shared one unmarked mode. So mark it. search() and sanitize_fts5_query() take an explicit allow_operators flag, default False — raw prose keeps the safe reading from the previous commit — and the harness, which knows it wrote operator syntax, opts in. Caught by tests/test_longmemeval_harness.py, which is why the arm assertion is worth keeping. --- benchmarking/longmemeval.py | 6 +++++- search_query.py | 11 +++++++++-- store.py | 7 +++++-- tests/test_lcm_core.py | 15 +++++++++++++++ 4 files changed, 34 insertions(+), 5 deletions(-) diff --git a/benchmarking/longmemeval.py b/benchmarking/longmemeval.py index 9053d68eb..1a57c6fab 100644 --- a/benchmarking/longmemeval.py +++ b/benchmarking/longmemeval.py @@ -463,7 +463,11 @@ def fts_hits(store, query: str, fetch: int) -> list[tuple[str, int]]: match_query = build_fts_query(query) if not match_query: return [] - rows = store.search(match_query, session_id=None, limit=fetch) + # build_fts_query composes FTS5 syntax on purpose (barewords joined by OR), + # so it opts in to operator handling. Raw prose never does. + rows = store.search( + match_query, session_id=None, limit=fetch, allow_operators=True + ) hits: list[tuple[str, int]] = [] for row in rows: store_id = row.get("store_id") diff --git a/search_query.py b/search_query.py index 2551a5ab6..83308767e 100644 --- a/search_query.py +++ b/search_query.py @@ -98,16 +98,23 @@ def _neutralize_bare_operators(sanitized: str) -> str: return "".join(out) -def sanitize_fts5_query(query: str) -> str: +def sanitize_fts5_query(query: str, *, allow_operators: bool = False) -> str: """Reduce a query to FTS5-safe terms, preserving balanced phrase quotes. Composed (NFC) first so a decomposed accent is one alphanumeric character rather than a base plus a combining mark, which is what unicode61 folds and indexes. The LIKE path deliberately does NOT normalize: it is a literal substring match against stored bytes, so it must not re-spell the query. + + ``allow_operators`` is the explicit marker for a query a CALLER composed as + FTS5 syntax (the benchmark harness joins its barewords with ``OR``). It + keeps bare AND/OR/NOT/NEAR intact. It must never be set for text that came + from a user or an agent: the default assumes raw prose, which is the only + safe reading when the two cannot be told apart. """ composed = unicodedata.normalize("NFC", query or "") - return _neutralize_bare_operators(_sanitize_query(composed, _fts5_safe_char)) + sanitized = _sanitize_query(composed, _fts5_safe_char) + return sanitized if allow_operators else _neutralize_bare_operators(sanitized) def sanitize_like_query(query: str) -> str: diff --git a/store.py b/store.py index 512a45991..6cba71e47 100644 --- a/store.py +++ b/store.py @@ -1098,7 +1098,8 @@ def search(self, query: str, session_id: str | None = None, conversation_id: str | None = None, role: str | None = None, time_from: float | None = None, - time_to: float | None = None) -> List[Dict[str, Any]]: + time_to: float | None = None, + allow_operators: bool = False) -> List[Dict[str, Any]]: """FTS5 search across raw messages. Retrieval contract: @@ -1109,8 +1110,10 @@ def search(self, query: str, session_id: str | None = None, - ``source='unknown'`` means the explicit unknown-source bucket, with legacy blank-source rows treated as equivalent for back-compat - ``conversation_id`` limits rows to one gateway conversation/session key + - ``allow_operators`` marks a query the CALLER composed as FTS5 syntax, + keeping its bare AND/OR/NOT/NEAR. Never set it for user or agent text """ - safe_query = sanitize_fts5_query(query) + safe_query = sanitize_fts5_query(query, allow_operators=allow_operators) terms = extract_search_terms(safe_query) phrases = extract_quoted_phrases(safe_query) # LIKE is the fallback for text sanitization LOSES (CJK/emoji) and for a diff --git a/tests/test_lcm_core.py b/tests/test_lcm_core.py index b8cb66c3f..1e91f7ca1 100644 --- a/tests/test_lcm_core.py +++ b/tests/test_lcm_core.py @@ -1922,6 +1922,21 @@ def test_search_does_not_let_a_raw_query_acquire_or_semantics(self, store): assert [row["store_id"] for row in results] == [both] + def test_search_honors_operators_only_when_the_caller_opts_in(self, store): + """The two modes are now marked: a caller that composed FTS5 syntax on + purpose (the benchmark harness joins barewords with OR) keeps its + disjunction; raw prose never acquires one.""" + alpha = store.append("sess1", {"role": "user", "content": "alpha only here"}) + beta = store.append("sess1", {"role": "user", "content": "beta only here"}) + + deliberate = store.search( + "alpha OR beta", session_id="sess1", allow_operators=True + ) + assert {row["store_id"] for row in deliberate} == {alpha, beta} + + raw = store.search("alpha OR beta", session_id="sess1") + assert raw == [] # conjunction of alpha, or, beta — nothing has all three + def test_search_survives_a_leading_boolean_operator(self, store): """`NOT ready` was an FTS syntax error that dumped the query on LIKE.""" store.append("sess1", {"role": "user", "content": "not ready for launch"}) From 169c3a3745b193e79f04f05c0a35f3cab5a15e9c Mon Sep 17 00:00:00 2001 From: EVA Date: Wed, 29 Jul 2026 03:53:07 +0700 Subject: [PATCH 29/54] review finding 2 (residual): release warmed matrices before a streamed scan MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit cache=False kept NEW batches out of the LRU but did nothing about what was already in it, and the retrieval-core pool keeps a store alive for the whole process. The delta-review probe measured cache_before=4 / cache_after=4 across a two-batch scan: four matrices warmed by earlier calls — ~192MB of summary float32 at a 25k batch, before the separate chunk cache — sitting resident alongside the streamed batch. The one-batch peak claim was false in exactly the long-lived-process case that motivates it. A multi-batch sweep now releases every cached matrix before allocating its first batch. All four caches go: both float32 LRUs and both binary-prescreen caches, since all of them are retained scan state and the binary ones are additive to the same peak. Same probe after the fix: cache_before=4, cache_after=0, and 0 resident at EVERY batch allocation. Safe on a shared pooled store: clearing only drops dict entries. A caller mid-scan holds its own reference to the tuple it was handed, so its matrix stays alive until it returns and is freed normally after. Single-batch scans are untouched and keep their warm-cache behavior, pinned by the existing test. --- DELTA-REVIEW-PACKET.md | 24 ++++++++++++++++ tests/test_vector_store.py | 58 ++++++++++++++++++++++++++++++++++++++ vector_store.py | 29 +++++++++++++++++++ 3 files changed, 111 insertions(+) create mode 100644 DELTA-REVIEW-PACKET.md diff --git a/DELTA-REVIEW-PACKET.md b/DELTA-REVIEW-PACKET.md new file mode 100644 index 000000000..49b8b8fe4 --- /dev/null +++ b/DELTA-REVIEW-PACKET.md @@ -0,0 +1,24 @@ +# DELTA REVIEW — PR #169 fix commits only (review ONLY, no modifications) + +Prior full review (sol·max) returned APPROVE-WITH-FIXES with 6 findings; the author fixed all six +(commits c13f4a2, 902fc37, c6fb1da, 4de8727, 530bd98+e4577f1, 5b58154 — one per finding, regression test each, +built from your repros). Scope: `git diff f960d9f..e4577f1` (the fixes only). The PR comment maps commits to +findings. Verify: + +1. Each finding actually closed by its commit (re-run your original repro logic mentally or via in-memory + SQLite probes; e.g. requires_like_fallback("art-related") now False; NFD naïve matches; "Portland, OR + hotel" has no operator semantics under the default mode). +2. **The mode split (finding 5's real fix):** `search(..., allow_operators=False)` default with harness + opt-in — sweep EVERY caller of search/sanitizer entry points: does any raw-user-query path get + allow_operators=True? Does any deliberate-FTS caller silently lose operators? (The author found + benchmarking/longmemeval.build_fts_query relies on OR — verify its opt-in is correct and no other caller + was missed.) +3. **Finding 2's fix:** multi-batch sweeps now stream past the LRU — verify peak memory is truly one batch + (no reference retention), single-batch paths still use the cache byte-identically, and no double-fetch. +4. The two flagged semantic changes: compound queries now conjunctive-on-index (was disjunctive-on-LIKE) — + any real caller for whom that is a regression? And the four retargeted LIKE-trigger tests — do they still + test their original contracts? +5. Any NEW hole opened by the fixes themselves. + +Output: VERDICT APPROVE / APPROVE-WITH-FIXES (mandatory list) / REJECT, findings with file:line + severity, +max 8, real defects only. No hardening suggestions. diff --git a/tests/test_vector_store.py b/tests/test_vector_store.py index f119a5911..c92d7181a 100644 --- a/tests/test_vector_store.py +++ b/tests/test_vector_store.py @@ -478,6 +478,64 @@ def test_multi_batch_scan_does_not_populate_or_thrash_the_matrix_cache(tmp_path) dag.close() +def test_multi_batch_scan_releases_matrices_warmed_before_it(tmp_path, monkeypatch): + """Delta-review residual on finding 2: cache=False keeps NEW batches out of + the LRU but leaves matrices warmed by EARLIER calls resident for the pooled + store's whole lifetime — the reviewer's probe measured cache_before=4 / + cache_after=4 across a two-batch scan, i.e. ~192MB coexisting with the + streamed batch. The invariant: at first-batch allocation, nothing is left. + """ + numpy = pytest.importorskip("numpy") + db_path = tmp_path / "cache-release.db" + dag = SummaryDAG(db_path) + store = VectorStore(db_path, bounded_scan_rows=2) # 2-row batches, 8 vectors + try: + _seed_scan_corpus( + dag, store, 8, gold_vector=[1.0, 0.0, 0.0], filler_vector=[0.0, 1.0, 0.0] + ) + identity = str(store._current_profile()["identity_hash"]) + ids = store._bounded_candidate_ids( + identity, + since=None, + until=None, + conversation_ids=None, + source=None, + limit=vector_store_module._SCAN_ALL_ROWS, + ) + # Warm the summary LRU to capacity the way earlier bounded calls would, + # and plant sentinels in the sibling caches that are additive to it. + for index in range(store._MATRIX_CACHE_MAX_ENTRIES): + store._numpy_rows(numpy, identity, 3, ids[index:index + 2]) + store._chunk_matrix_cache[("sentinel", 0, ())] = ([], [], [], None) + store._binary_matrix_cache[("sentinel", 0)] = ([], None) + store._chunk_binary_matrix_cache[("sentinel", 0)] = ([], None) + cache_before = len(store._matrix_cache) + assert cache_before == store._MATRIX_CACHE_MAX_ENTRIES + + observed: list[tuple[int, int]] = [] + original = store._load_matrix + + def probe(np, identity_hash, dim, embedded_ids, dtype): + observed.append((len(store._matrix_cache), len(store._chunk_matrix_cache))) + return original(np, identity_hash, dim, embedded_ids, dtype) + + monkeypatch.setattr(store, "_load_matrix", probe) + result = store.knn([1.0, 0.0, 0.0], k=1, model="scan", full_scan=True) + + assert result.coverage == "full" + assert len(observed) == 4 # 8 vectors, 2 per batch + # Released BEFORE the first batch is allocated, and never repopulated. + assert observed[0] == (0, 0) + assert all(sizes == (0, 0) for sizes in observed) + assert len(store._matrix_cache) == 0 + assert len(store._chunk_matrix_cache) == 0 + assert len(store._binary_matrix_cache) == 0 + assert len(store._chunk_binary_matrix_cache) == 0 + finally: + store.close() + dag.close() + + def test_single_batch_scan_still_caches_its_matrix(tmp_path): """The warm pooled-store path is unchanged when one batch covers the corpus.""" db_path = tmp_path / "cache-single.db" diff --git a/vector_store.py b/vector_store.py index 49ad504bf..19ce60a1e 100644 --- a/vector_store.py +++ b/vector_store.py @@ -1488,6 +1488,29 @@ def _ranked( for _, embedded_id, score, kind in ranked[:limit] ] + def _release_matrix_caches(self) -> None: + """Drop every cached matrix before a streamed scan allocates its batches. + + ``cache=False`` alone keeps NEW batches out of the LRU but does nothing + about what is already in it, and the retrieval-core pool keeps a store + alive for the whole process. A probe over a two-batch scan measured + cache_before=4 / cache_after=4: four warm matrices (~192MB of summary + float32 at a 25k batch, before the separate chunk cache) coexisting with + the streamed batch, which is not the one-batch bound this batching + promises. Both float32 caches and both binary-prescreen caches are + released, since all four are retained scan state. + + Safe against a concurrent reader on a shared pooled store: clearing only + drops the dict entries. A caller mid-scan holds its own reference to the + tuple it was handed, so its matrix stays alive until it returns and is + freed normally afterwards. + """ + with self._cache_lock: + self._matrix_cache.clear() + self._chunk_matrix_cache.clear() + self._binary_matrix_cache.clear() + self._chunk_binary_matrix_cache.clear() + def _scan_ranked( self, *, @@ -1523,6 +1546,12 @@ def _scan_ranked( scanned = 0 stopped_early = False cache_batches = len(candidate_ids) <= batch_rows + if not cache_batches: + # Keeping the new batches OUT of the cache is only half the bound: + # matrices warmed by EARLIER calls stay resident for the pooled + # store's whole lifetime. Release them before the first batch is + # allocated, so peak really is one batch. + self._release_matrix_caches() started = time.monotonic() for start in range(0, len(candidate_ids), batch_rows): batch = candidate_ids[start:start + batch_rows] From 18ba0a59191656e0fc19423ed329248398370d7e Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Thu, 23 Jul 2026 12:26:39 +0700 Subject: [PATCH 30/54] embed: configurable generous query-path spend guard (#123) The query embedding path (resolve_provider(for_backfill=False)) implicitly used the strict default EmbeddingSpendGuard() (60 calls/60s/60s backoff), a backfill-economy control that silently gutted retrieval once a tight query loop crossed 60 calls/window -- every further call was rejected pre-network with ProviderRateLimited (451 query embeds is negligible spend). Make the query-path guard explicit and configurable via embedding_query_spend_* (env LCM_EMBEDDING_QUERY_SPEND_*), defaulting to a generous 600/60s/60s. Backfill keeps its exempt bulk contract (max_calls=0). From 1b45d6108dff8e47348eecb3332fbf11ada87d01 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Thu, 23 Jul 2026 12:28:39 +0700 Subject: [PATCH 31/54] test: query-path spend guard exemption + typed rate-limit reason (#123) - query path generous/unthrottled at 120 back-to-back embeds (root-cause repro) - guard configurable via constructor args and LCM_EMBEDDING_QUERY_SPEND_* env - backfill path stays exempt (max_calls=0) -- bulk contract unchanged - zero-discard: a tripped guard surfaces the TYPED ProviderRateLimited with the exact probe-recorded message, pre-network, never a swallowed counter Bounded 429 wait-retry (spec item b) is already covered by test_voyage_429_honors_retry_after_and_caps_budget. From afbb000ce7c402f645d42e10a59c8f0976243fe8 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 04:03:10 +0700 Subject: [PATCH 32/54] fix: preserve store_id on summary recall hits (#164) --- tests/test_lcm_recall.py | 39 +++++++++++++++++++++++++++++++++++++-- tools.py | 8 ++++++++ 2 files changed, 45 insertions(+), 2 deletions(-) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index 34a411730..d61717648 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -64,7 +64,15 @@ def recall_engine(tmp_path): store.close() -def _add_summary(engine, summary, *, session_id, created_at, latest_at=None): +def _add_summary( + engine, + summary, + *, + session_id, + created_at, + latest_at=None, + source_ids=None, +): return engine._dag.add_node( SummaryNode( session_id=session_id, @@ -72,7 +80,7 @@ def _add_summary(engine, summary, *, session_id, created_at, latest_at=None): summary=summary, token_count=20, source_token_count=40, - source_ids=[], + source_ids=list(source_ids or []), source_type="messages", created_at=created_at, earliest_at=created_at, @@ -206,6 +214,33 @@ def test_recall_returns_cross_session_summaries_without_a_filter(recall_engine, assert payload["provenance"]["arms_run"] == ["summary"] +def test_summary_hit_carries_direct_source_store_id(recall_engine, monkeypatch): + store_id = recall_engine._store.append( + "session-a", + {"role": "user", "content": "kanban dashboard sprint source row"}, + ) + node_id = _add_summary( + recall_engine, + "kanban dashboard sprint summary", + session_id="session-a", + created_at=10.0, + source_ids=[store_id], + ) + _seed_summary_vectors(recall_engine, [(node_id, [1.0, 0.0])]) + + payload = _recall( + recall_engine, + monkeypatch, + include="summaries", + scope_bias=0.0, + limit=5, + ) + + hit = next(hit for hit in payload["hits"] if hit["node_id"] == node_id) + assert hit["kind"] == "summary" + assert hit["store_id"] == store_id + + def test_scope_bias_boosts_current_conversation_without_filtering(recall_engine, monkeypatch): cross = _add_summary(recall_engine, "cross conversation kanban", session_id="session-a", created_at=5.0) here = _add_summary(recall_engine, "current conversation kanban", session_id=CURRENT, created_at=5.0) diff --git a/tools.py b/tools.py index b8289e86f..59780979a 100644 --- a/tools.py +++ b/tools.py @@ -3762,9 +3762,15 @@ def _lcm_recall_summary_arm( current = engine.current_session_id hits: list[dict[str, Any]] = [] for node, _score in nodes: + source_store_id = ( + int(node.source_ids[0]) + if node.source_type == "messages" and node.source_ids + else None + ) hit = { "kind": "summary", "node_id": node.node_id, + "store_id": source_store_id, "session_id": node.session_id, "timestamp": node.latest_at or node.created_at or 0, "snippet": (node.summary or "")[:_LCM_RECALL_SNIPPET_CHARS], @@ -4247,6 +4253,8 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: } if hit.get("kind") == "summary": item["node_id"] = hit.get("node_id") + if hit.get("store_id") is not None: + item["store_id"] = hit.get("store_id") else: item["store_id"] = hit.get("store_id") if hit.get("chunk_span"): From b16f39b5f981199c4ca9d50143e252363940a230 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 04:04:44 +0700 Subject: [PATCH 33/54] chore: prepare v0.20.0 R2 train --- .github/ISSUE_TEMPLATE/bug_report.yml | 2 +- CHANGELOG.md | 8 ++++++++ README.md | 2 +- docs/operator-guide.md | 2 +- plugin.yaml | 2 +- tests/test_lcm_command.py | 4 ++-- tests/test_lcm_engine.py | 10 +++++----- tests/test_packaging_install.py | 2 +- 8 files changed, 20 insertions(+), 12 deletions(-) diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml index 1cdadedf2..023f09b81 100644 --- a/.github/ISSUE_TEMPLATE/bug_report.yml +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -44,7 +44,7 @@ body: attributes: label: Version and branch description: hermes-lcm version, branch, commit, or release tag. - placeholder: v0.19.0, main, or commit SHA + placeholder: v0.20.0, main, or commit SHA validations: required: true diff --git a/CHANGELOG.md b/CHANGELOG.md index e7a54c827..5fbd0e380 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,14 @@ This repo also publishes GitHub Releases. This file is the repo-root release sur ## Unreleased +## v0.20.0 - 2026-07-29 + +Release focus: benchmark-driven retrieval scaling, evidence provenance, and bounded query embedding spend. + +- Removed the large-corpus recall ceilings by scanning the full summary and chunk corpora in bounded batches and sanitizing raw natural-language FTS queries before fallback. (#169) +- Made the query-path embedding spend guard configurable with a generous default while preserving the exempt backfill contract. (stephenschoettler/hermes-lcm#434) +- Preserved a direct source `store_id` on summary recall hits so strict evidence renderers can validate their source identity. (#164) + ## v0.19.0 - 2026-07-07 Release focus: data-safety hardening, operator diagnostics, import tooling, benchmarking, and the WS5 engine decomposition. diff --git a/README.md b/README.md index a0477f664..4a23ae02a 100644 --- a/README.md +++ b/README.md @@ -205,7 +205,7 @@ Typical output: ```text Plugins (1): - ✓ hermes-lcm v0.19.0 (8 tools) + ✓ hermes-lcm v0.20.0 (8 tools) Provider Plugins: Context Engine: lcm diff --git a/docs/operator-guide.md b/docs/operator-guide.md index 2c3cc2701..d063fc5bb 100644 --- a/docs/operator-guide.md +++ b/docs/operator-guide.md @@ -99,7 +99,7 @@ Typical output: ```text Plugins (1): - ✓ hermes-lcm v0.19.0 (8 tools) + ✓ hermes-lcm v0.20.0 (8 tools) Provider Plugins: Context Engine: lcm diff --git a/plugin.yaml b/plugin.yaml index 5e6e6358a..8ab314364 100644 --- a/plugin.yaml +++ b/plugin.yaml @@ -1,5 +1,5 @@ name: hermes-lcm -version: 0.19.0 +version: 0.20.0 description: "Lossless Context Management — standalone Hermes plugin with a DAG-based context engine that never loses a message" author: "Voltropy / Hermes Community" provides_tools: diff --git a/tests/test_lcm_command.py b/tests/test_lcm_command.py index 02e40a18b..abb082c83 100644 --- a/tests/test_lcm_command.py +++ b/tests/test_lcm_command.py @@ -417,7 +417,7 @@ def test_lcm_status_reports_runtime_identity(engine): repo_root = Path(__file__).resolve().parent.parent assert "plugin_name: hermes-lcm" in result - assert "plugin_version: 0.19.0" in result + assert "plugin_version: 0.20.0" in result assert f"plugin_path: {repo_root}" in result assert "module_path:" in result assert "database_path_source: config.database_path" in result @@ -457,7 +457,7 @@ def test_lcm_doctor_reports_health_checks(engine): assert "messages_fts: ok" in result assert "nodes_fts: ok" in result assert "plugin_name: hermes-lcm" in result - assert "plugin_version: 0.19.0" in result + assert "plugin_version: 0.20.0" in result assert f"plugin_path: {repo_root}" in result assert "plugin_git_commit:" in result assert "triage_guidance:\n- none" in result diff --git a/tests/test_lcm_engine.py b/tests/test_lcm_engine.py index 61d2ae8f1..048c95611 100644 --- a/tests/test_lcm_engine.py +++ b/tests/test_lcm_engine.py @@ -1139,7 +1139,7 @@ def test_get_status_exposes_runtime_identity_for_loaded_plugin_tree(tmp_path): assert identity["engine"] == "lcm" assert identity["plugin_name"] == "hermes-lcm" - assert identity["plugin_version"] == "0.19.0" + assert identity["plugin_version"] == "0.20.0" assert Path(identity["plugin_path"]) == repo_root assert Path(identity["module_path"]).name == "engine.py" assert Path(identity["database_path"]) == db_path @@ -1165,11 +1165,11 @@ def test_plugin_metadata_refreshes_when_manifest_changes(tmp_path, monkeypatch): initial = identity_mod._plugin_metadata() assert initial["name"] == "hermes-lcm" - assert initial["version"] == "0.19.0" + assert initial["version"] == "0.20.0" - updated = original.replace('version: "0.19.0"', 'version: "9.9.9-test"') + updated = original.replace('version: "0.20.0"', 'version: "9.9.9-test"') if updated == original: - updated = original.replace('version: 0.19.0', 'version: 9.9.9-test') + updated = original.replace('version: 0.20.0', 'version: 9.9.9-test') assert updated != original try: @@ -1207,7 +1207,7 @@ def test_lcm_doctor_json_includes_runtime_identity(engine): payload = json.loads(engine.handle_tool_call("lcm_doctor", {})) assert payload["runtime_identity"]["plugin_name"] == "hermes-lcm" - assert payload["runtime_identity"]["plugin_version"] == "0.19.0" + assert payload["runtime_identity"]["plugin_version"] == "0.20.0" assert "plugin_git_commit" in payload["runtime_identity"] diff --git a/tests/test_packaging_install.py b/tests/test_packaging_install.py index f05455de2..31fe1f93c 100644 --- a/tests/test_packaging_install.py +++ b/tests/test_packaging_install.py @@ -377,7 +377,7 @@ def test_plugin_entrypoint_registers_lcm_context_engine(): identity = engine.get_status()["runtime_identity"] repo_root = Path(__file__).resolve().parent.parent assert identity["plugin_name"] == "hermes-lcm" - assert identity["plugin_version"] == "0.19.0" + assert identity["plugin_version"] == "0.20.0" assert Path(identity["plugin_path"]) == repo_root assert identity["database_path_source"] in {"config.database_path", "hermes_home", "default_home"} assert identity["plugin_git_commit"] From a3a47e9091ad342fb1373e0b5896e6328d1356c0 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 05:43:57 +0700 Subject: [PATCH 34/54] =?UTF-8?q?recall:=20reference-strict=20answer=5Frea?= =?UTF-8?q?dy=20delivery=20=E2=80=94=20never=20deliver=20an=20uncitable=20?= =?UTF-8?q?hit=20(#164/F35)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The #168 sanitization fix woke the summary arm's internal FTS queries, so kind:"summary" hits went 2/100 -> 16/100 questions on the V1 sanity slice. Every delivered summary hit is uncitable: #164a's store_id path populated ZERO on real stores, and consumers that validate evidence references fail CLOSED on an unreferenced card, so one uncitable hit destroys the whole response (16/100 questions lost, F32 baseline 2). The invariant: in reference-strict delivery no hit lacking a validated (store_id, char_start, char_end) source span is delivered; an omitted hit is backfilled by the next-ranked citable one, so the delivered count stays at LIMIT, and the omission count is surfaced in provenance.answer_ready. Mechanism: the rule filters the RANKED CANDIDATE list before session-diversity selection, so backfill is structural rather than a second pass. Citability is decided at the slot a candidate would occupy — inside the hydration budget it resolves via content_offset, later it must bring its own chunk_span. A post-hydration guard on the built item is a fail-closed backstop for a missing store row. With nothing uncitable the filter is the identity, so delivery is byte-identical. Summaries are ranking signal, not citable evidence, and are not given a reference: a node's summary is model-generated prose, not a verbatim span of any row, so lcm::- would assert bytes that are not there. SummaryNode.source_ids lists EVERY message a leaf node summarizes, so #164a's source_ids[0] named lineage, not a citation — unsound where it fired, not merely incomplete. Scope: detail='answer_ready' only (the citation-bearing mode); the default 'snippets' response makes no source-span claim and is untouched. ON by default — an adapter-set flag would leave every other caller on the same contract still receiving uncitable evidence — with LCMConfig.recall_reference_strict=False as the opt-out for a host that renders recall without citations. The four answer_ready mechanics tests that exercise diversity/hydration/cap using summary hits are pinned to the disabled path, where delivery stays byte-identical. --- config.py | 11 +++ tests/test_lcm_recall.py | 17 +++++ tools.py | 155 ++++++++++++++++++++++++++++++++++++--- 3 files changed, 174 insertions(+), 9 deletions(-) diff --git a/config.py b/config.py index 34fdc5643..29755fa49 100644 --- a/config.py +++ b/config.py @@ -374,6 +374,7 @@ class _EnvFieldSpec: _EnvFieldSpec("recall_scan_rows", "LCM_RECALL_SCAN_ROWS", int), _EnvFieldSpec("recall_scan_max_rows", "LCM_RECALL_SCAN_MAX_ROWS", int), _EnvFieldSpec("recall_scan_budget_s", "LCM_RECALL_SCAN_BUDGET_S", float), + _EnvFieldSpec("recall_reference_strict", "LCM_RECALL_REFERENCE_STRICT", bool), _EnvFieldSpec("proactive_recall_enabled", "LCM_PROACTIVE_RECALL_ENABLED", bool), _EnvFieldSpec("proactive_recall_min_score", "LCM_PROACTIVE_RECALL_MIN_SCORE", float), _EnvFieldSpec("proactive_recall_budget_tokens", "LCM_PROACTIVE_RECALL_BUDGET_TOKENS", int), @@ -659,6 +660,16 @@ class LCMConfig: recall_arm_weights: dict[str, float] = field( default_factory=lambda: dict(_DEFAULT_RECALL_ARM_WEIGHTS) ) + # Reference-strict delivery for detail='answer_ready' (FINDING-F35 §2): a hit + # that cannot carry a truthful (store_id, char_start, char_end) source span is + # never delivered as evidence; the next-ranked citable hit takes its slot and + # the omission count is surfaced in provenance.answer_ready. ON by default -- + # delivering evidence the product cannot cite is a correctness defect, and the + # consumers that validate references fail CLOSED on an unreferenced card, so a + # single uncitable hit destroys the whole response. Set False only for a host + # that renders recall without citations and wants summary hits back; disabled, + # the answer_ready response is byte-identical to the pre-F35 delivery. + recall_reference_strict: bool = True # -- Proactive memory injection (SPEC F, default-OFF) --- # At active-context assembly, embed the newest user message and run the # lcm_recall pipeline to surface cross-session memories the model would diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index d61717648..f37217b92 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -147,6 +147,19 @@ def _summary_hit(engine, node_id): } +def _non_strict(engine): + """Pin a case to the pre-F35 delivery by disabling reference-strict. + + These cases exercise answer_ready MECHANICS (session diversity, the + hydration budget, the response cap) using summary hits, which the default + reference-strict delivery no longer hands out as evidence. Running them on + the disabled path keeps them exercising the mechanic they were written for + and doubles as the byte-identity guarantee for the opt-out. + """ + engine._config.recall_reference_strict = False + return engine + + def _patch_summary_arm(monkeypatch, hits): monkeypatch.setattr( lcm_tools, @@ -664,6 +677,7 @@ def test_recall_reports_query_embedding_provider_and_usage(recall_engine, monkey def test_answer_ready_applies_stable_post_rank_session_diversity( recall_engine, monkeypatch ): + _non_strict(recall_engine) node_ids = [] for index in range(7): node_ids.append( @@ -768,6 +782,7 @@ def test_answer_ready_centers_message_content_on_exact_chunk_span( def test_answer_ready_expands_summary_ref_with_2400_char_bound( recall_engine, monkeypatch ): + _non_strict(recall_engine) summary = "kanban dashboard sprint " + "summary-evidence " * 240 node = _add_summary( recall_engine, @@ -798,6 +813,7 @@ def test_answer_ready_expands_summary_ref_with_2400_char_bound( def test_answer_ready_expands_only_first_eight_and_reports_policy( recall_engine, monkeypatch ): + _non_strict(recall_engine) node_ids = [ _add_summary( recall_engine, @@ -843,6 +859,7 @@ def test_answer_ready_expands_only_first_eight_and_reports_policy( def test_answer_ready_enforces_complete_response_cap_and_marks_query_truncation( recall_engine, monkeypatch ): + _non_strict(recall_engine) node = _add_summary( recall_engine, "bounded summary evidence", diff --git a/tools.py b/tools.py index 59780979a..3f92e290a 100644 --- a/tools.py +++ b/tools.py @@ -3513,6 +3513,104 @@ def _lcm_recall_diverse_entries( return selected, dropped +def _lcm_recall_reference_shape(hit: dict[str, Any], *, hydratable: bool) -> str | None: + """Name the delivery shape that gives this hit a truthful source reference. + + Reference-strict delivery (FINDING-F35 §2). Exactly two shapes reach the + caller with a mechanically checkable ``(store_id, char_start, char_end)`` + span into a real stored row: + + * ``content_offset`` -- a hydrated message excerpt carries the exact window + it read (``store_id`` + ``content_offset`` + ``content_returned_chars``); + * ``chunk_span`` -- a message hit outside the hydration budget carries the + verbatim chunk span it was retrieved from. + + A summary has NEITHER, and cannot be given one. Its text is model-generated + prose, not a verbatim span of any row, so ``lcm::-`` + would assert bytes that are not at that offset. ``SummaryNode.source_ids`` + is the list of *every* message a leaf node summarizes, so even a + message-sourced node's first source is lineage, not a citation (#164a). + Summaries stay RANKING SIGNAL -- they fuse and order, they are not delivered + as evidence. + """ + if hit.get("kind") == "summary": + return None + if hit.get("store_id") is None: + return None + if hydratable: + return "content_offset" + if hit.get("chunk_span"): + return "chunk_span" + return None + + +def _lcm_recall_item_is_referenced(item: dict[str, Any]) -> bool: + """Post-hydration check that a BUILT item really carries its source span. + + The candidate filter admits an in-budget message hit on the promise that + hydration will attach ``content_offset``/``content_returned_chars``. A store + row that has gone missing breaks that promise, so the finished item is + re-checked against what it actually carries rather than what was predicted. + """ + if item.get("kind") == "summary" or item.get("store_id") is None: + return False + if isinstance(item.get("exact_ref"), str) and item["exact_ref"]: + return True + if item.get("content_offset") is not None and item.get("content_returned_chars"): + return True + return bool(item.get("chunk_span")) + + +def _lcm_recall_citable_entries( + ordered: list[dict[str, Any]], + *, + limit: int, + per_session_limit: int, + expanded_limit: int, +) -> tuple[list[dict[str, Any]], int, int]: + """Reference-strict variant of :func:`_lcm_recall_diverse_entries`. + + Same stable rank-preserving selection with bounded session density, with one + added admission rule: a candidate that cannot carry a validated source + reference at the slot it would occupy is skipped. Because the rule filters + the RANKED CANDIDATE LIST rather than the finished result, the selection + simply continues down the ranking -- an omitted hit is backfilled by the + next-ranked citable one and the delivered count stays at ``limit``. + + Citability depends on the slot: the first ``expanded_limit`` selections are + inside the hydration budget and resolve via ``content_offset``; later ones + must bring their own ``chunk_span``. The reference check runs BEFORE the + session-density check so an undelivered hit never consumes session quota. + When nothing is uncitable this is the identity -- same selection, same + ``diversity_dropped`` -- so delivery stays byte-identical. + """ + selected: list[dict[str, Any]] = [] + session_counts: dict[str, int] = {} + dropped = 0 + unreferenced = 0 + for entry in ordered: + hit = entry["hit"] + if _lcm_recall_reference_shape( + hit, hydratable=len(selected) < expanded_limit + ) is None: + unreferenced += 1 + continue + raw_session_id = hit.get("session_id") + session_key = ( + str(raw_session_id) + if raw_session_id not in {None, ""} + else f"missing:{_hit_identity(hit)!r}" + ) + if session_counts.get(session_key, 0) >= per_session_limit: + dropped += 1 + continue + session_counts[session_key] = session_counts.get(session_key, 0) + 1 + selected.append(entry) + if len(selected) >= limit: + break + return selected, dropped, unreferenced + + def _lcm_recall_content_window( content: Any, *, @@ -3932,6 +4030,13 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: if detail not in _LCM_RECALL_VALID_DETAIL: return json.dumps({"error": "detail must be one of: snippets, answer_ready"}) + # Reference-strict delivery applies to the citation-bearing mode only: the + # default 'snippets' response makes no source-span claim and carries no + # hydration, so strictness there would drop evidence for no reference gain. + reference_strict = detail == "answer_ready" and bool( + getattr(engine._config, "recall_reference_strict", True) + ) + delta_requested = "seen_refs" in args raw_seen_refs = args.get("seen_refs") if delta_requested and detail != "answer_ready": @@ -4204,21 +4309,36 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: # -- Response shaping (char-capped). The default snippets path retains the # historical order and serialized response exactly. answer_ready applies # stable post-rank diversity before bounded exact-ref hydration. + unreferenced_dropped = 0 if detail == "answer_ready": - selected_entries, diversity_dropped = _lcm_recall_diverse_entries( - ordered, - limit=_LCM_RECALL_LIMIT_CAP if delta_requested else limit, - per_session_limit=_LCM_RECALL_ANSWER_READY_PER_SESSION_LIMIT, + selection_limit = _LCM_RECALL_LIMIT_CAP if delta_requested else limit + expanded_limit = ( + _LCM_RECALL_LIMIT_CAP + if delta_requested + else _LCM_RECALL_ANSWER_READY_EXPANDED_HIT_LIMIT ) + if reference_strict: + ( + selected_entries, + diversity_dropped, + unreferenced_dropped, + ) = _lcm_recall_citable_entries( + ordered, + limit=selection_limit, + per_session_limit=_LCM_RECALL_ANSWER_READY_PER_SESSION_LIMIT, + expanded_limit=expanded_limit, + ) + else: + selected_entries, diversity_dropped = _lcm_recall_diverse_entries( + ordered, + limit=selection_limit, + per_session_limit=_LCM_RECALL_ANSWER_READY_PER_SESSION_LIMIT, + ) answer_ready_content = _lcm_recall_answer_ready_content( engine, selected_entries, query=query, - expanded_limit=( - _LCM_RECALL_LIMIT_CAP - if delta_requested - else _LCM_RECALL_ANSWER_READY_EXPANDED_HIT_LIMIT - ), + expanded_limit=expanded_limit, ) if delta_requested: selected_entries = [ @@ -4238,6 +4358,7 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: hits_out: list[dict[str, Any]] = [] response_chars = 0 response_cap_truncated = False + unreferenced_omitted = 0 for entry in selected_entries: hit = entry["hit"] arms = sorted({arm_order[index] for index in entry["ranks"].keys()}) @@ -4302,6 +4423,13 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: else "ingest_fallback" ), } + # Fail-closed backstop: the candidate filter admitted this hit on the + # promise of a hydrated window, so only a missing store row can land + # here. Omit rather than deliver an uncitable card; the count is + # disclosed instead of the shortfall being silent. + if reference_strict and not _lcm_recall_item_is_referenced(item): + unreferenced_omitted += 1 + continue item_chars = len(json.dumps(item, ensure_ascii=False)) if hits_out and response_chars + item_chars > _LCM_RECALL_RESPONSE_CHAR_CAP: response_cap_truncated = True @@ -4365,6 +4493,15 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: "hydration_policy": "bounded exact reads only; no additional retrieval search", "response_truncated": response_cap_truncated, } + if reference_strict: + expansion["reference_strict"] = True + expansion["unreferenced_dropped_count"] = unreferenced_dropped + expansion["unreferenced_omitted_count"] = unreferenced_omitted + expansion["reference_policy"] = ( + "no hit lacking a validated (store_id, char_start, char_end) source " + "span is delivered; dropped candidates are backfilled by the " + "next-ranked citable hit, so summaries rank but never cite" + ) response["detail"] = detail response["provenance"]["detail"] = detail response["provenance"]["answer_ready"] = expansion From 511d93eb01920ca4b5043b94dfbb18e5c06872f1 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 06:10:09 +0700 Subject: [PATCH 35/54] test: pin the reference-strict delivery invariant (#164/F35) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Unit coverage for the candidate filter and end-to-end coverage through lcm_recall: * all three uncitable shapes lose their slot to the next-ranked citable hit (nested summary, #164a message-sourced summary, and a message that reached a post-hydration slot without a chunk span of its own) — delivered count still reaches LIMIT; * slot position decides citability: a chunk-span-less message is admitted while the hydration budget still covers it and rejected once it falls past it; * an undelivered hit never consumes per-session density quota; * a mixed hit set delivers messages only, backfilled to LIMIT, with the omissions counted in provenance.answer_ready; * include='summaries' + detail='answer_ready' returns an empty result plus the count rather than uncitable evidence; * the detail='snippets' default still delivers summaries untouched; * flag-off inertness: with the opt-out set, the reference-strict functions are monkeypatched to raise and delivery still completes on the legacy path with none of the new provenance keys — byte-identical by construction. The two end-to-end cases silence the FTS arm so summary-arm and chunk-arm ranks interleave; seeded messages match lexically too, and that extra arm would lift them clear above the summary hits, leaving the uncitable candidates below the cut and never exercising the backfill. --- tests/test_lcm_recall.py | 298 +++++++++++++++++++++++++++++++++++++++ 1 file changed, 298 insertions(+) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index f37217b92..c89cc7c59 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -1386,3 +1386,301 @@ def rerank(self, query, documents, *, top_k=None, timeout, model="rerank-2.5-lit # Had the 0.9 voyage score been spliced onto the RRF scale it would dwarf the # ~0.016 RRF score; the reported score must stay RRF-scaled. assert all(hit["score"] < 0.1 for hit in payload["hits"]) + + +# -- Reference-strict delivery (FINDING-F35) ---------------------------------- +# +# The #168 sanitization fix woke the summary arm's internal FTS queries, so +# uncitable kind:"summary" hits started reaching consumers that validate every +# evidence reference and fail CLOSED on an unreferenced card. These cases pin +# the invariant: no hit lacking a validated (store_id, char_start, char_end) +# source span is delivered, an omitted hit is backfilled by the next-ranked +# citable one, and the omission count is surfaced rather than silent. + + +def _message_entry(store_id, *, session_id="session-a", chunk_span=None): + hit = { + "kind": "message_excerpt", + "store_id": store_id, + "session_id": session_id, + } + if chunk_span is not None: + hit["chunk_span"] = chunk_span + return {"hit": hit} + + +def _summary_entry(node_id, *, session_id="session-s", store_id=None): + hit = {"kind": "summary", "node_id": node_id, "session_id": session_id} + if store_id is not None: + hit["store_id"] = store_id + return {"hit": hit} + + +def test_reference_strict_backfills_past_every_uncitable_candidate_shape(): + """The three uncitable shapes lose their slot to the next citable hit. + + Candidates 1 and 4 are uncitable summaries (the second carries the #164a + ``store_id``, which names lineage rather than a citation); candidate 6 is a + message that reached a post-hydration slot with no chunk span of its own. + All three are skipped and the selection continues down the ranking, so the + delivered count still reaches ``limit``. + """ + span = {"char_start": 0, "char_end": 40} + ordered = [ + _summary_entry(901), + _message_entry(1, chunk_span=span), + _message_entry(2, chunk_span=span), + _summary_entry(902, store_id=7), + _message_entry(3, chunk_span=span), + _message_entry(4), # no chunk span, and slot 3 is past expanded_limit=2 + _message_entry(5, chunk_span=span), + _message_entry(6, chunk_span=span), + ] + + selected, dropped, unreferenced = lcm_tools._lcm_recall_citable_entries( + ordered, + limit=5, + per_session_limit=5, + expanded_limit=2, + ) + + assert [entry["hit"]["store_id"] for entry in selected] == [1, 2, 3, 5, 6] + assert len(selected) == 5 + assert unreferenced == 3 + assert dropped == 0 + + +def test_reference_strict_admits_an_unspanned_message_inside_the_hydration_budget(): + """Slot position decides: a chunk-span-less message is citable while the + hydration budget still covers it (it will carry content_offset), and only + becomes uncitable once it falls past that budget.""" + ordered = [_message_entry(index) for index in range(1, 5)] + + selected, _dropped, unreferenced = lcm_tools._lcm_recall_citable_entries( + ordered, + limit=4, + per_session_limit=5, + expanded_limit=2, + ) + + assert [entry["hit"]["store_id"] for entry in selected] == [1, 2] + assert unreferenced == 2 + + +def test_reference_strict_skips_uncitable_before_it_consumes_session_quota(): + """An undelivered hit must not spend the per-session density budget it was + never going to occupy.""" + ordered = [_summary_entry(900 + index, session_id="session-a") for index in range(3)] + ordered += [_message_entry(index, session_id="session-a") for index in range(1, 6)] + + selected, dropped, unreferenced = lcm_tools._lcm_recall_citable_entries( + ordered, + limit=5, + per_session_limit=5, + expanded_limit=8, + ) + + assert [entry["hit"]["store_id"] for entry in selected] == [1, 2, 3, 4, 5] + assert unreferenced == 3 + assert dropped == 0 + + +def _only_vector_arms(monkeypatch): + """Silence the FTS arm so summary-arm and chunk-arm ranks interleave. + + Seeded messages match the query lexically too, and that extra arm lifts them + clear above the summary hits — which would leave the uncitable candidates + below the cut and never exercise the backfill. + """ + monkeypatch.setattr(lcm_tools, "_lcm_recall_fts_arm", lambda *_a, **_k: ([], None)) + + +def _seed_citable_messages(engine, count, *, sessions=("session-a", "session-b")): + """Seed messages that the chunk arm can retrieve with a verbatim span.""" + match = "kanban dashboard sprint" + store_ids = [] + for index in range(count): + content = f"{match} evidence body {index} " + "filler " * 20 + store_id = engine._store.append( + sessions[index % len(sessions)], + {"role": "user", "content": content}, + source="chat", + ) + store_ids.append(store_id) + _seed_chunk_vectors(engine, [(store_id, 0, 0, len(content), [1.0, 0.0])]) + return store_ids + + +def test_reference_strict_delivers_only_citable_hits_and_reports_the_omissions( + recall_engine, monkeypatch +): + """End-to-end: a mixed hit set delivers messages only, backfilled to LIMIT, + with the uncitable summaries counted in provenance meta.""" + _only_vector_arms(monkeypatch) + store_ids = _seed_citable_messages(recall_engine, 10) + node_ids = [ + _add_summary( + recall_engine, + f"kanban dashboard sprint rollup {index}", + session_id="session-s", + created_at=10.0, + latest_at=time.time(), + ) + for index in range(3) + ] + _patch_summary_arm( + monkeypatch, + [_summary_hit(recall_engine, node_id) for node_id in node_ids], + ) + + payload = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + scope_bias=0.0, + limit=6, + ) + + hits = payload["hits"] + assert len(hits) == 6, "omitted summaries must be backfilled, not lost" + assert {hit["kind"] for hit in hits} == {"message_excerpt"} + assert all(hit["store_id"] in store_ids for hit in hits) + # Every delivered hit resolves to a truthful span in a real stored row. + assert all( + lcm_tools._lcm_recall_item_is_referenced(hit) for hit in hits + ) + policy = payload["provenance"]["answer_ready"] + assert policy["reference_strict"] is True + assert policy["unreferenced_dropped_count"] == len(node_ids) + assert policy["unreferenced_omitted_count"] == 0 + + +def test_reference_strict_drops_the_164a_message_sourced_summary_too( + recall_engine, monkeypatch +): + """#164a populated store_id from ``source_ids[0]``. A leaf node's source_ids + lists EVERY message it summarizes and its text is generated prose, so that + store_id is lineage, not a citation — the hit stays undelivered.""" + _only_vector_arms(monkeypatch) + store_ids = _seed_citable_messages(recall_engine, 4) + node = _add_summary( + recall_engine, + "kanban dashboard sprint rollup", + session_id="session-s", + created_at=10.0, + latest_at=time.time(), + source_ids=store_ids, + ) + summary_hit = _summary_hit(recall_engine, node) + summary_hit["store_id"] = store_ids[0] + _patch_summary_arm(monkeypatch, [summary_hit]) + + payload = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + scope_bias=0.0, + limit=4, + ) + + assert all(hit["kind"] == "message_excerpt" for hit in payload["hits"]) + assert payload["provenance"]["answer_ready"]["unreferenced_dropped_count"] == 1 + + +def test_reference_strict_include_summaries_returns_nothing_rather_than_uncitable( + recall_engine, monkeypatch +): + """include='summaries' with the citation-bearing detail asks for citable + delivery of non-citable material. The honest answer is an empty result plus + the omission count — not a carve-out that reinstates the defect.""" + node_ids = [ + _add_summary( + recall_engine, + f"kanban dashboard sprint rollup {index}", + session_id=f"session-{index}", + created_at=10.0, + ) + for index in range(3) + ] + _patch_summary_arm( + monkeypatch, + [_summary_hit(recall_engine, node_id) for node_id in node_ids], + ) + + payload = _recall( + recall_engine, + monkeypatch, + include="summaries", + detail="answer_ready", + scope_bias=0.0, + limit=5, + ) + + assert payload["hits"] == [] + assert payload["total_results"] == 0 + assert payload["provenance"]["answer_ready"]["unreferenced_dropped_count"] == 3 + + +def test_reference_strict_leaves_the_snippets_default_untouched( + recall_engine, monkeypatch +): + """The default detail makes no source-span claim and carries no hydration, + so strictness must not cost it evidence.""" + node = _add_summary( + recall_engine, + "kanban dashboard sprint rollup", + session_id="session-a", + created_at=10.0, + ) + _seed_summary_vectors(recall_engine, [(node, [1.0, 0.0])]) + + payload = _recall(recall_engine, monkeypatch, include="summaries", limit=5) + + assert [hit["node_id"] for hit in payload["hits"]] == [node] + assert "answer_ready" not in payload["provenance"] + + +def test_reference_strict_disabled_never_enters_the_new_delivery_path( + recall_engine, monkeypatch +): + """Flag-off inertness: with the opt-out set, none of the reference-strict + code runs and the response carries none of its provenance keys, so delivery + is byte-identical to the pre-F35 path by construction.""" + _non_strict(recall_engine) + _seed_citable_messages(recall_engine, 4) + node_ids = [ + _add_summary( + recall_engine, + f"kanban dashboard sprint rollup {index}", + session_id="session-s", + created_at=10.0, + latest_at=time.time(), + ) + for index in range(2) + ] + _patch_summary_arm( + monkeypatch, + [_summary_hit(recall_engine, node_id) for node_id in node_ids], + ) + + def _forbidden(*_args, **_kwargs): + raise AssertionError("reference-strict code ran on the disabled path") + + monkeypatch.setattr(lcm_tools, "_lcm_recall_citable_entries", _forbidden) + monkeypatch.setattr(lcm_tools, "_lcm_recall_item_is_referenced", _forbidden) + + payload = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + scope_bias=0.0, + limit=6, + ) + + # The legacy path still delivers the (uncitable) summaries it always did. + assert any(hit["kind"] == "summary" for hit in payload["hits"]) + policy = payload["provenance"]["answer_ready"] + assert "reference_strict" not in policy + assert "unreferenced_dropped_count" not in policy + assert "unreferenced_omitted_count" not in policy + assert "reference_policy" not in policy From 0aab863d941277a917fc6f483de418ffea9fb173 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 06:49:41 +0700 Subject: [PATCH 36/54] recall: carry summary-KNN relevance onto source messages (review finding 1) Defect: reference-strict delivery did not merely stop delivering summaries, it silently amputated the whole arm. RRF keys a summary by node and a message by store_id, so with rerank off -- the default -- dropping summary entries after fusion leaves every message score and the entire order exactly as if the summary arm had never run. A session reachable ONLY by summary KNN contributed no evidence at all, and adaptive retrieval's summary-lead path (which forces answer_ready and then extracts summary handles as non-evidence leads) received none. The PR's claim that "summaries fuse and order" was false in the default configuration. Repro: a store whose gold session shares no query term with its messages and has no chunk vectors, reachable only through a matching summary vector, returned nothing citable in strict mode. Fix: the arm emits ordinary message_excerpt candidates for the rows beneath each ranked node, in node-rank order, resolved through a new SummaryDAG.source_message_ids() that walks source_ids down through nested nodes to their messages leaves. They are NOT the summary wearing a message's content -- the anti-pattern PR #173's review flagged: each is a real row with its own identity, verbatim excerpt and truthful content_offset, fused by RRF against the FTS and chunk arms and competing on merit rather than inheriting the node's slot. Generated summary prose is still never delivered. The node handles come back as provenance.answer_ready.summary_leads -- locators only, no summary text. _extract_search_leads walks the whole payload for locator keys, so this restores the adaptive drill-down path without putting prose back into evidence. Bounded fan-out (4 rows/node): inside a node there is no per-message relevance signal -- that is what the FTS and chunk arms are for -- so the slice is deterministic and the arm stays inside its candidate budget. The purpose is to make the SESSION reachable with citable evidence, not to rank within it. --- dag.py | 43 +++++++++++++++++ tests/test_lcm_recall.py | 88 ++++++++++++++++++++++++++++++++-- tools.py | 101 +++++++++++++++++++++++++++++++++++++-- 3 files changed, 226 insertions(+), 6 deletions(-) diff --git a/dag.py b/dag.py index f4abe0509..700c0a890 100644 --- a/dag.py +++ b/dag.py @@ -718,6 +718,49 @@ def get_source_nodes(self, node: SummaryNode) -> List[SummaryNode]: ).fetchall() return [self._row_to_node(r) for r in rows] + def source_message_ids(self, node_id: int, *, limit: int) -> List[int]: + """Resolve a node to the store_ids of the messages underneath it. + + Walks ``source_ids`` down through nested nodes to the ``messages`` leaves, + so a derived (depth > 0) node resolves to real rows rather than to the + child nodes it was built from. Ordered by store_id and bounded by + ``limit`` so a node summarizing a long session cannot flood a caller. + + A node's own summary text is generated prose and is never a citation for + these rows; this is the lineage link, used to let the source MESSAGES be + retrieved and cited in their own right. + """ + if limit <= 0: + return [] + with self._db_lock: + rows = self._conn.execute( + """ + WITH RECURSIVE source_walk(source_type, source_id) AS ( + SELECT n.source_type, CAST(j.value AS INTEGER) + FROM summary_nodes n, json_each(n.source_ids) j + WHERE n.node_id = ? + + UNION + + SELECT child.source_type, CAST(j.value AS INTEGER) + FROM summary_nodes child + JOIN source_walk walk + ON walk.source_type = 'nodes' + AND child.node_id = walk.source_id + JOIN json_each(child.source_ids) j + ) + SELECT DISTINCT m.store_id + FROM source_walk walk + JOIN messages m + ON walk.source_type = 'messages' + AND m.store_id = walk.source_id + ORDER BY m.store_id + LIMIT ? + """, + (node_id, limit), + ).fetchall() + return [int(row[0]) for row in rows] + def _node_matches_source( self, node_id: int, diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index c89cc7c59..d5a826805 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -164,7 +164,7 @@ def _patch_summary_arm(monkeypatch, hits): monkeypatch.setattr( lcm_tools, "_lcm_recall_summary_arm", - lambda *_args, **_kwargs: (list(hits), "full", len(hits), len(hits)), + lambda *_args, **_kwargs: (list(hits), "full", len(hits), len(hits), []), ) @@ -496,8 +496,8 @@ def test_answer_ready_delta_is_opt_in_and_returns_only_novel_exact_refs( "session-b", {"role": "user", "content": "kanban dashboard sprint beta"} ) monkeypatch.setattr(lcm_tools, "resolve_provider", lambda _config: MockProvider()) - monkeypatch.setattr(lcm_tools, "_lcm_recall_summary_arm", lambda *_a, **_k: ([], "none", 0, 0)) - monkeypatch.setattr(lcm_tools, "_lcm_recall_chunk_arm", lambda *_a, **_k: ([], "none", 0, 0)) + monkeypatch.setattr(lcm_tools, "_lcm_recall_summary_arm", lambda *_a, **_k: ([], "none", 0, 0, [])) + monkeypatch.setattr(lcm_tools, "_lcm_recall_chunk_arm", lambda *_a, **_k: ([], "none", 0, 0, [])) primary = _recall( recall_engine, @@ -1684,3 +1684,85 @@ def _forbidden(*_args, **_kwargs): assert "unreferenced_dropped_count" not in policy assert "unreferenced_omitted_count" not in policy assert "reference_policy" not in policy + + +# -- Cross-model review of PR #174 (four mandatory P2 findings) --------------- + + +def test_summary_only_session_still_yields_citable_evidence_in_strict_mode( + recall_engine, monkeypatch +): + """FINDING 1: strict mode must not amputate the summary arm. + + RRF keys a summary by node and a message by store_id, so dropping summary + entries after fusion leaves message scores and order exactly as if the arm + had never run. Here the gold session is reachable ONLY by summary + similarity -- its messages share no query term and have no chunk vectors -- + so if the arm's influence were lost the session would contribute nothing. + """ + store_ids = [ + recall_engine._store.append( + "session-gold", + {"role": "user", "content": f"the quarterly budget reconciliation note {index}"}, + source="chat", + ) + for index in range(3) + ] + node = _add_summary( + recall_engine, + "kanban dashboard sprint rollup", + session_id="session-gold", + created_at=10.0, + source_ids=store_ids, + ) + _seed_summary_vectors(recall_engine, [(node, [1.0, 0.0])]) + + payload = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + scope_bias=0.0, + limit=5, + ) + + hits = payload["hits"] + assert hits, "the summary-only session must still reach the caller" + assert {hit["kind"] for hit in hits} == {"message_excerpt"} + assert {hit["store_id"] for hit in hits} <= set(store_ids) + # Real rows, cited truthfully -- not the node's generated prose. + assert all(hit["session_id"] == "session-gold" for hit in hits) + assert all(hit["content_offset"] == 0 for hit in hits) + assert "kanban dashboard sprint rollup" not in json.dumps(hits) + + +def test_summary_nodes_come_back_as_non_evidence_leads(recall_engine, monkeypatch): + """FINDING 1: adaptive retrieval's summary-lead path keeps working. + + ``_extract_search_leads`` walks the whole tool payload for locator handles, + so surfacing the node ids in provenance restores the drill-down path that + dropping summary hits would otherwise have closed -- without putting + generated prose back into evidence. + """ + node = _add_summary( + recall_engine, + "kanban dashboard sprint rollup", + session_id="session-gold", + created_at=10.0, + source_ids=[ + recall_engine._store.append( + "session-gold", {"role": "user", "content": "budget note"}, source="chat" + ) + ], + ) + _seed_summary_vectors(recall_engine, [(node, [1.0, 0.0])]) + + payload = _recall( + recall_engine, monkeypatch, detail="answer_ready", scope_bias=0.0, limit=5 + ) + + leads = payload["provenance"]["answer_ready"]["summary_leads"] + assert [lead["node_id"] for lead in leads] == [node] + assert leads[0]["session_id"] == "session-gold" + # A locator, never the summary text. + assert "summary" not in leads[0] + assert "snippet" not in leads[0] diff --git a/tools.py b/tools.py index 3f92e290a..bf042a045 100644 --- a/tools.py +++ b/tools.py @@ -284,6 +284,11 @@ def _parse_strict_int(value: Any, name: str) -> tuple[int | None, str | None]: _LCM_RECALL_VALID_DETAIL = frozenset({"snippets", "answer_ready"}) _LCM_RECALL_ANSWER_READY_PER_SESSION_LIMIT = 5 _LCM_RECALL_ANSWER_READY_EXPANDED_HIT_LIMIT = 8 +# Bounded per-node fan-out when reference-strict delivery carries summary-KNN +# relevance onto the source messages beneath a ranked node. Small on purpose: +# the point is to make the SESSION reachable with citable evidence, and the FTS +# and chunk arms are what rank individual messages inside it. +_LCM_RECALL_SUMMARY_SOURCE_PER_NODE = 4 _LCM_RECALL_ANSWER_READY_CONTENT_CHARS = 2_400 # Recency boost half-life (30 days) and its floor: a memory's rank_score is # multiplied by 2**(-age/half_life), clamped so age never zeroes an otherwise @@ -3816,6 +3821,84 @@ def _lcm_recall_scan_bounds(engine: "LCMEngine") -> dict[str, Any]: } +def _lcm_recall_summary_source_hits( + engine: "LCMEngine", + nodes: list[tuple[Any, float]], + *, + current: str | None, + candidate_limit: int, +) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + """Carry summary-KNN relevance onto the nodes' SOURCE MESSAGES. + + Reference-strict delivery cannot hand out a summary, and simply dropping the + hits would silently amputate the arm: RRF keys a summary by node and a + message by store_id, so with rerank off (the default) removing summary + entries after fusion leaves every message score and the whole order exactly + as if the arm had never run -- a session reachable ONLY by summary KNN would + contribute no evidence at all. + + So the arm emits ordinary ``message_excerpt`` candidates for the rows beneath + each ranked node, in node-rank order. They are NOT the summary wearing a + message's content: each is a real row with its own identity, its own verbatim + excerpt and its own truthful ``content_offset``, fused by RRF against the FTS + and chunk arms like any other candidate and competing on merit rather than + inheriting the node's slot. The node handles come back separately as + non-evidence leads. + + Bounded fan-out: within a node there is no per-message relevance signal (that + is what the FTS and chunk arms are for), so a deterministic ``store_id``- + ordered slice per node is taken and the arm stays inside its candidate + budget. The purpose is to make the SESSION reachable with citable evidence, + not to rank inside it. + """ + leads: list[dict[str, Any]] = [] + ordered_ids: list[int] = [] + seen: set[int] = set() + for node, _score in nodes: + lead: dict[str, Any] = {"node_id": node.node_id, "session_id": node.session_id} + hint = _lcm_recall_summary_expand_hint({"session_id": node.session_id}) + if hint: + lead["expand_hint"] = hint + leads.append(lead) + if len(ordered_ids) >= candidate_limit: + continue + for store_id in engine._dag.source_message_ids( + node.node_id, limit=_LCM_RECALL_SUMMARY_SOURCE_PER_NODE + ): + if store_id in seen: + continue + seen.add(store_id) + ordered_ids.append(store_id) + + hits: list[dict[str, Any]] = [] + if ordered_ids: + rows = engine._store.get_batch(ordered_ids[:candidate_limit]) + for store_id in ordered_ids[:candidate_limit]: + row = rows.get(store_id) + if row is None: + continue + content = str(row.get("content") or "") + if not content: + continue + session_id = row.get("session_id") + hit = { + "kind": "message_excerpt", + "store_id": store_id, + "session_id": session_id, + "source": row.get("source") or "", + "role": row.get("role"), + "timestamp": row.get("timestamp") or 0, + # The excerpt is the row's own prefix, so offset 0 is the truth + # here rather than the fabricated default a consumer would guess. + "content_offset": 0, + "snippet": content[:_LCM_RECALL_SNIPPET_CHARS], + "from_current_session": bool(current) and session_id == current, + } + hit["expand_hint"] = _lcm_recall_excerpt_expand_hint(hit) + hits.append(hit) + return hits, leads + + def _lcm_recall_summary_arm( engine: "LCMEngine", *, @@ -3823,6 +3906,7 @@ def _lcm_recall_summary_arm( provider: Any, candidate_limit: int, deadline: float, + reference_strict: bool = False, ) -> tuple[list[dict[str, Any]], str]: """Summary KNN arm: embedded summaries across ALL sessions (no filter).""" knn_results = _run_within_deadline( @@ -3846,7 +3930,7 @@ def _lcm_recall_summary_arm( coverage = knn_results.coverage ranked_rows = list(knn_results) if coverage == "none" or not ranked_rows: - return [], coverage, knn_results.scanned, knn_results.total + return [], coverage, knn_results.scanned, knn_results.total, [] nodes = _run_within_deadline( lambda: hydrate_semantic_nodes( engine, @@ -3858,6 +3942,11 @@ def _lcm_recall_summary_arm( name="lcm-recall-summary-hydrate", ) current = engine.current_session_id + if reference_strict: + source_hits, leads = _lcm_recall_summary_source_hits( + engine, nodes, current=current, candidate_limit=candidate_limit + ) + return source_hits, coverage, knn_results.scanned, knn_results.total, leads hits: list[dict[str, Any]] = [] for node, _score in nodes: source_store_id = ( @@ -3876,7 +3965,7 @@ def _lcm_recall_summary_arm( } hit["expand_hint"] = _lcm_recall_summary_expand_hint(hit) hits.append(hit) - return hits, coverage, knn_results.scanned, knn_results.total + return hits, coverage, knn_results.scanned, knn_results.total, [] def _lcm_recall_chunk_arm( @@ -4068,6 +4157,10 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: run_chunk = include in {"all", "verbatim"} arm_hits: dict[str, list[dict[str, Any]]] = {} + # Node handles from the summary arm. Reference-strict delivery routes summary + # relevance through the source messages, so the nodes themselves are surfaced + # here as non-evidence drill-down leads instead of as hits. + summary_leads: list[dict[str, Any]] = [] coverage: dict[str, str] = {} degraded_reasons: list[str] = [] embedding_query_metrics: list[dict[str, Any]] = [] @@ -4175,12 +4268,13 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: if query_vector is not None: if run_summary: try: - hits, cov, scanned, total = _lcm_recall_summary_arm( + hits, cov, scanned, total, summary_leads = _lcm_recall_summary_arm( engine, query_vector=query_vector, provider=provider, candidate_limit=candidate_limit, deadline=deadline, + reference_strict=reference_strict, ) arm_hits["summary"] = hits coverage["summary"] = cov @@ -4497,6 +4591,7 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: expansion["reference_strict"] = True expansion["unreferenced_dropped_count"] = unreferenced_dropped expansion["unreferenced_omitted_count"] = unreferenced_omitted + expansion["summary_leads"] = summary_leads expansion["reference_policy"] = ( "no hit lacking a validated (store_id, char_start, char_end) source " "span is delivered; dropped candidates are backfilled by the " From 6f501f5b2e931891010b5712194721b8ff480feb Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 06:50:20 +0700 Subject: [PATCH 37/54] recall: validate and publish every delivered span (review findings 2, 3, 4) One mechanism closes three findings, because they are the same defect seen from three sides: admission predicted a reference instead of establishing one. Splitting them would produce intermediate commits that assert an invariant they do not yet enforce -- notably, emitting an offset (finding 2) without validating it (finding 4) would ship exactly the fabrication both findings are about. FINDING 2 -- post-hydration hits omitted the public offset. Hits past slot 8 were admitted on a chunk_span but the response exposed no content_offset or exact_ref, so __init__.py's _answer_ready_baseline substituted offset 0 and constructed a FALSE reference for every excerpt not starting at its row's beginning. In this PR's own fixed replay all 272 post-slot-8 hits omitted the offset and 19 had nonzero chunk starts across 11/16 qids (concrete: qid 37f165cf, store 431, chunk start 2217). The pinned renderer passed only because it reads chunk_span, masking the in-repo consumer failure. Every delivered hit now publishes the (content_offset, content_returned_chars) span its text actually occupies. The rendered reference is unchanged -- the renderer's content_offset branch yields the same string its chunk_span branch did -- so prompt bytes do not move. FINDING 4 -- a stale chunk_span could become a claimed strict reference. No coherent snapshot spans chunk hydration and response shaping, so a row deleted or rewritten in between (store.py:506 delete_session_messages) would be delivered with a dangling span. Truth is now established by reading the row: the delivered bytes must be present at the claimed offset of the CURRENT row. Hydration already read the row for the in-budget slots, so one batched read covers the rest -- never a query per hit. The same check also rejects an FTS snippet, which is a match window with markers rather than a verbatim prefix, so its content_offset of 0 was never a real location. FINDING 3 -- a hydration miss could not backfill. Selection stopped dead at limit, so a row that vanished between the ranking and hydration reads left the response short while citable candidates sat unexamined just below the cut. The selector is now resumable: the caller draws a replacement and the ranked walk picks up where it left off with the same rules and session ledger, counters advancing only over what was examined. A run needing no replacement walks exactly as far as before and reports the identical diversity_dropped. Regression tests, each mutation-checked to fail without its mechanism: the true chunk offset is published rather than 0; no admitted hit lacks a public offset; a deletion between the reads keeps the count at limit; a row rewritten after chunk hydration never yields a reference-strict item. --- tests/test_lcm_recall.py | 166 +++++++++++++++++++++- tools.py | 287 ++++++++++++++++++++++++++++----------- 2 files changed, 366 insertions(+), 87 deletions(-) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index d5a826805..bd06eeb2d 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -497,7 +497,7 @@ def test_answer_ready_delta_is_opt_in_and_returns_only_novel_exact_refs( ) monkeypatch.setattr(lcm_tools, "resolve_provider", lambda _config: MockProvider()) monkeypatch.setattr(lcm_tools, "_lcm_recall_summary_arm", lambda *_a, **_k: ([], "none", 0, 0, [])) - monkeypatch.setattr(lcm_tools, "_lcm_recall_chunk_arm", lambda *_a, **_k: ([], "none", 0, 0, [])) + monkeypatch.setattr(lcm_tools, "_lcm_recall_chunk_arm", lambda *_a, **_k: ([], "none", 0, 0)) primary = _recall( recall_engine, @@ -1545,9 +1545,12 @@ def test_reference_strict_delivers_only_citable_hits_and_reports_the_omissions( assert len(hits) == 6, "omitted summaries must be backfilled, not lost" assert {hit["kind"] for hit in hits} == {"message_excerpt"} assert all(hit["store_id"] in store_ids for hit in hits) - # Every delivered hit resolves to a truthful span in a real stored row. + # Every delivered hit PUBLISHES the span its text occupies, so a consumer + # never has to guess an offset. assert all( - lcm_tools._lcm_recall_item_is_referenced(hit) for hit in hits + isinstance(hit.get("content_offset"), int) + and hit.get("content_returned_chars") + for hit in hits ) policy = payload["provenance"]["answer_ready"] assert policy["reference_strict"] is True @@ -1666,8 +1669,9 @@ def test_reference_strict_disabled_never_enters_the_new_delivery_path( def _forbidden(*_args, **_kwargs): raise AssertionError("reference-strict code ran on the disabled path") - monkeypatch.setattr(lcm_tools, "_lcm_recall_citable_entries", _forbidden) - monkeypatch.setattr(lcm_tools, "_lcm_recall_item_is_referenced", _forbidden) + monkeypatch.setattr(lcm_tools, "_LcmRecallStrictSelector", _forbidden) + monkeypatch.setattr(lcm_tools, "_lcm_recall_verified_span", _forbidden) + monkeypatch.setattr(lcm_tools, "_lcm_recall_strict_rows", _forbidden) payload = _recall( recall_engine, @@ -1766,3 +1770,155 @@ def test_summary_nodes_come_back_as_non_evidence_leads(recall_engine, monkeypatc # A locator, never the summary text. assert "summary" not in leads[0] assert "snippet" not in leads[0] + + +def test_admitted_hit_publishes_the_true_chunk_offset_not_zero( + recall_engine, monkeypatch +): + """FINDING 2: a post-hydration hit must publish its real offset. + + ``__init__.py``'s ``_answer_ready_baseline`` substitutes ``content_offset`` + 0 when the field is absent, which fabricates a reference for every excerpt + that does not start at the beginning of its row. + """ + match = "kanban dashboard sprint" + content = "a" * 2_217 + match + "z" * 500 + store_id = recall_engine._store.append( + "session-a", {"role": "user", "content": content}, source="chat" + ) + _seed_chunk_vectors( + recall_engine, [(store_id, 0, 2_217, len(content), [1.0, 0.0])] + ) + # Push it past the hydration budget so it is admitted on its chunk span. + filler = _seed_citable_messages(recall_engine, 9, sessions=("session-b", "session-c")) + _only_vector_arms(monkeypatch) + + payload = _recall( + recall_engine, + monkeypatch, + include="verbatim", + detail="answer_ready", + scope_bias=0.0, + limit=25, + ) + + hit = next(h for h in payload["hits"] if h["store_id"] == store_id) + assert "content" not in hit, "this case must exercise the un-hydrated path" + assert hit["content_offset"] == 2_217 + assert hit["content_returned_chars"] == len(hit["snippet"]) + # The published span is where the text really is. + start = hit["content_offset"] + assert content[start:start + hit["content_returned_chars"]] == hit["snippet"] + assert filler + + +def test_every_admitted_hit_publishes_a_public_offset(recall_engine, monkeypatch): + """FINDING 2: no admitted hit may leave the offset for a consumer to guess.""" + _only_vector_arms(monkeypatch) + _seed_citable_messages(recall_engine, 12) + + payload = _recall( + recall_engine, + monkeypatch, + include="verbatim", + detail="answer_ready", + scope_bias=0.0, + limit=12, + ) + + assert payload["hits"] + for hit in payload["hits"]: + assert isinstance(hit["content_offset"], int) + assert hit["content_returned_chars"] > 0 + + +def test_hydration_miss_refills_from_the_ranked_tail(recall_engine, monkeypatch): + """FINDING 3: a row deleted between the reads must not underfill. + + Selection stops at ``limit``; if a selected row then vanishes, omitting it + without drawing a replacement leaves the response short while citable + candidates sit unexamined just below the cut. + """ + _only_vector_arms(monkeypatch) + _seed_citable_messages(recall_engine, 12) + + # Doom rows the run actually DELIVERS, so the refill is genuinely exercised + # rather than the deletion landing on candidates below the cut. + baseline = _recall( + recall_engine, + monkeypatch, + include="verbatim", + detail="answer_ready", + scope_bias=0.0, + limit=8, + ) + assert len(baseline["hits"]) == 8 + doomed = {hit["store_id"] for hit in baseline["hits"][:2]} + real_get_batch = recall_engine._store.get_batch + + def deleting_get_batch(ids): + # Simulate delete_session_messages landing between the ranking read and + # the hydration read: the rows are simply not there any more. + return { + key: value + for key, value in real_get_batch(ids).items() + if key not in doomed + } + + monkeypatch.setattr(recall_engine._store, "get_batch", deleting_get_batch) + + payload = _recall( + recall_engine, + monkeypatch, + include="verbatim", + detail="answer_ready", + scope_bias=0.0, + limit=8, + ) + + assert len(payload["hits"]) == 8, "the count must survive a hydration miss" + assert not (doomed & {hit["store_id"] for hit in payload["hits"]}) + assert payload["total_results"] == 8 + + +def test_stale_chunk_span_never_becomes_a_strict_reference( + recall_engine, monkeypatch +): + """FINDING 4: a chunk index is not proof the row still says that. + + Cleanup can delete or rewrite the message after chunk hydration and before + response shaping; without a check against the current row the stale hit + would be delivered as reference-strict with a dangling span. + """ + _only_vector_arms(monkeypatch) + content = "kanban dashboard sprint " + "evidence " * 40 + store_id = recall_engine._store.append( + "session-a", {"role": "user", "content": content}, source="chat" + ) + _seed_chunk_vectors(recall_engine, [(store_id, 0, 0, len(content), [1.0, 0.0])]) + other = _seed_citable_messages(recall_engine, 9, sessions=("session-b", "session-c")) + real_get_batch = recall_engine._store.get_batch + + def rewriting_get_batch(ids): + rows = dict(real_get_batch(ids)) + if store_id in rows: + # The row now holds different bytes than the chunk index recorded. + rows[store_id] = {**rows[store_id], "content": "unrelated replacement"} + return rows + + monkeypatch.setattr(recall_engine._store, "get_batch", rewriting_get_batch) + + payload = _recall( + recall_engine, + monkeypatch, + include="verbatim", + detail="answer_ready", + scope_bias=0.0, + limit=25, + ) + + delivered = {hit["store_id"] for hit in payload["hits"]} + assert store_id not in delivered, "a stale span must not be delivered as strict" + assert delivered <= set(other) + for hit in payload["hits"]: + assert isinstance(hit["content_offset"], int) diff --git a/tools.py b/tools.py index bf042a045..afd2c0a20 100644 --- a/tools.py +++ b/tools.py @@ -3521,22 +3521,20 @@ def _lcm_recall_diverse_entries( def _lcm_recall_reference_shape(hit: dict[str, Any], *, hydratable: bool) -> str | None: """Name the delivery shape that gives this hit a truthful source reference. - Reference-strict delivery (FINDING-F35 §2). Exactly two shapes reach the - caller with a mechanically checkable ``(store_id, char_start, char_end)`` - span into a real stored row: - - * ``content_offset`` -- a hydrated message excerpt carries the exact window - it read (``store_id`` + ``content_offset`` + ``content_returned_chars``); - * ``chunk_span`` -- a message hit outside the hydration budget carries the - verbatim chunk span it was retrieved from. - - A summary has NEITHER, and cannot be given one. Its text is model-generated - prose, not a verbatim span of any row, so ``lcm::-`` - would assert bytes that are not at that offset. ``SummaryNode.source_ids`` - is the list of *every* message a leaf node summarizes, so even a - message-sourced node's first source is lineage, not a citation (#164a). - Summaries stay RANKING SIGNAL -- they fuse and order, they are not delivered - as evidence. + Reference-strict delivery (FINDING-F35 §2). This is only the CHEAP, + rank-ordered ADMISSION test -- it says a candidate could plausibly resolve to + a ``(store_id, char_start, char_end)`` span, not that it does. Truth is + established later by :func:`_lcm_recall_verified_span`, which reads the row + and checks that the delivered text really sits at the claimed offset. + + A summary is rejected here and cannot be given a reference. Its text is + model-generated prose, not a verbatim span of any row, so + ``lcm::-`` would assert bytes that are not at that + offset. ``SummaryNode.source_ids`` is the list of *every* message a leaf node + summarizes, so even a message-sourced node's first source is lineage, not a + citation (#164a). The summary arm keeps its ranking influence through + :func:`_lcm_recall_summary_source_hits`, which lets the nodes' SOURCE + MESSAGES compete as ordinary citable candidates. """ if hit.get("kind") == "summary": return None @@ -3546,24 +3544,76 @@ def _lcm_recall_reference_shape(hit: dict[str, Any], *, hydratable: bool) -> str return "content_offset" if hit.get("chunk_span"): return "chunk_span" + # An arm that already knows where its excerpt sits (the chunk arm's + # char_start, a summary-source hit's row prefix) can be cited without + # hydration -- subject to the verification below. + if hit.get("content_offset") is not None: + return "content_offset" return None -def _lcm_recall_item_is_referenced(item: dict[str, Any]) -> bool: - """Post-hydration check that a BUILT item really carries its source span. +def _lcm_recall_verified_span( + item: dict[str, Any], row: dict[str, Any] | None +) -> tuple[int, int] | None: + """Return the ``(offset, chars)`` the delivered text ACTUALLY occupies. + + The one honest definition of a validated source reference: the bytes handed + to the caller must be present at the claimed offset of the CURRENT row. That + single check subsumes three failure modes a structural test misses -- - The candidate filter admits an in-budget message hit on the promise that - hydration will attach ``content_offset``/``content_returned_chars``. A store - row that has gone missing breaks that promise, so the finished item is - re-checked against what it actually carries rather than what was predicted. + * a stale ``chunk_span`` whose row was deleted or rewritten between chunk + hydration and response shaping (no coherent snapshot spans those reads); + * an FTS snippet, which is a match window with markers rather than a + verbatim prefix, so its ``content_offset`` of 0 is not a real location; + * a hydration miss, where the promised window never arrived. + + Returning the span rather than a bool is deliberate: the caller PUBLISHES it + (``content_offset``/``content_returned_chars``) so consumers do not have to + re-derive it. ``__init__.py``'s ``_answer_ready_baseline`` substitutes offset + 0 when the field is absent, which silently fabricates a reference for every + hit whose excerpt does not start at the beginning of its row. """ if item.get("kind") == "summary" or item.get("store_id") is None: - return False - if isinstance(item.get("exact_ref"), str) and item["exact_ref"]: - return True - if item.get("content_offset") is not None and item.get("content_returned_chars"): - return True - return bool(item.get("chunk_span")) + return None + text = item.get("content") if item.get("content") is not None else item.get("snippet") + text = str(text or "") + if not text: + return None + raw_offset = item.get("content_offset") + if raw_offset is None: + return None + try: + offset = int(raw_offset) + except (TypeError, ValueError, OverflowError): + return None + if offset < 0 or row is None: + return None + content = str(row.get("content") or "") + if content[offset:offset + len(text)] != text: + return None + return offset, len(text) + + +def _lcm_recall_strict_rows( + engine: "LCMEngine", + entries: list[dict[str, Any]], + answer_ready_content: dict[tuple, dict[str, Any]], +) -> dict[int, dict[str, Any]]: + """Read the current rows the un-hydrated admitted candidates claim to quote. + + Hydration already read the row for every in-budget slot and cut the window + out of it, so those spans are true by construction. The candidates past that + budget were admitted on a chunk index alone -- one batched read (never a + query per hit) gives them the same standard of proof. + """ + store_ids = [ + int(entry["hit"]["store_id"]) + for entry in entries + if entry["hit"].get("kind") != "summary" + and entry["hit"].get("store_id") is not None + and _hit_identity(entry["hit"]) not in answer_ready_content + ] + return engine._store.get_batch(store_ids) if store_ids else {} def _lcm_recall_citable_entries( @@ -3573,47 +3623,82 @@ def _lcm_recall_citable_entries( per_session_limit: int, expanded_limit: int, ) -> tuple[list[dict[str, Any]], int, int]: - """Reference-strict variant of :func:`_lcm_recall_diverse_entries`. + """Reference-strict selection with no reserve (kept for direct unit use).""" + selector = _LcmRecallStrictSelector( + ordered, per_session_limit=per_session_limit, expanded_limit=expanded_limit + ) + selected = selector.take(limit) + return selected, selector.diversity_dropped, selector.unreferenced_dropped + + +class _LcmRecallStrictSelector: + """Reference-strict variant of :func:`_lcm_recall_diverse_entries`, resumable. Same stable rank-preserving selection with bounded session density, with one - added admission rule: a candidate that cannot carry a validated source - reference at the slot it would occupy is skipped. Because the rule filters - the RANKED CANDIDATE LIST rather than the finished result, the selection + added admission rule: a candidate that cannot plausibly carry a validated + source reference at the slot it would occupy is skipped. Because the rule + filters the RANKED CANDIDATE LIST rather than the finished result, selection simply continues down the ranking -- an omitted hit is backfilled by the next-ranked citable one and the delivered count stays at ``limit``. - Citability depends on the slot: the first ``expanded_limit`` selections are - inside the hydration budget and resolve via ``content_offset``; later ones - must bring their own ``chunk_span``. The reference check runs BEFORE the - session-density check so an undelivered hit never consumes session quota. - When nothing is uncitable this is the identity -- same selection, same - ``diversity_dropped`` -- so delivery stays byte-identical. + The walk is RESUMABLE because admission is only a prediction: a candidate can + still fail verification against its current row once the response is being + shaped (a deleted row, a stale chunk span). Stopping dead at ``limit`` would + then underfill while citable candidates sat unexamined just below the cut, so + the caller draws a replacement with :meth:`take` and the walk picks up where + it left off -- same rules, same session ledger, counters advanced only over + what was actually examined. A run that needs no replacement therefore walks + exactly as far as the non-resumable version did and reports the identical + ``diversity_dropped``; with nothing uncitable at all it is the identity, so + delivery stays byte-identical. + + The reference check runs BEFORE the session-density check so a candidate that + is never delivered cannot consume session quota. """ - selected: list[dict[str, Any]] = [] - session_counts: dict[str, int] = {} - dropped = 0 - unreferenced = 0 - for entry in ordered: - hit = entry["hit"] - if _lcm_recall_reference_shape( - hit, hydratable=len(selected) < expanded_limit - ) is None: - unreferenced += 1 - continue - raw_session_id = hit.get("session_id") - session_key = ( - str(raw_session_id) - if raw_session_id not in {None, ""} - else f"missing:{_hit_identity(hit)!r}" - ) - if session_counts.get(session_key, 0) >= per_session_limit: - dropped += 1 - continue - session_counts[session_key] = session_counts.get(session_key, 0) + 1 - selected.append(entry) - if len(selected) >= limit: - break - return selected, dropped, unreferenced + + def __init__( + self, + ordered: list[dict[str, Any]], + *, + per_session_limit: int, + expanded_limit: int, + ) -> None: + self._ordered = ordered + self._cursor = 0 + self._per_session_limit = per_session_limit + self._expanded_limit = expanded_limit + self._session_counts: dict[str, int] = {} + self._admitted = 0 + self.diversity_dropped = 0 + self.unreferenced_dropped = 0 + + def take(self, count: int) -> list[dict[str, Any]]: + """Admit up to ``count`` further candidates, resuming the ranked walk.""" + taken: list[dict[str, Any]] = [] + while self._cursor < len(self._ordered) and len(taken) < count: + entry = self._ordered[self._cursor] + self._cursor += 1 + hit = entry["hit"] + if _lcm_recall_reference_shape( + hit, hydratable=self._admitted < self._expanded_limit + ) is None: + self.unreferenced_dropped += 1 + continue + raw_session_id = hit.get("session_id") + session_key = ( + str(raw_session_id) + if raw_session_id not in {None, ""} + else f"missing:{_hit_identity(hit)!r}" + ) + if self._session_counts.get(session_key, 0) >= self._per_session_limit: + self.diversity_dropped += 1 + continue + self._session_counts[session_key] = ( + self._session_counts.get(session_key, 0) + 1 + ) + self._admitted += 1 + taken.append(entry) + return taken def _lcm_recall_content_window( @@ -4403,7 +4488,7 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: # -- Response shaping (char-capped). The default snippets path retains the # historical order and serialized response exactly. answer_ready applies # stable post-rank diversity before bounded exact-ref hydration. - unreferenced_dropped = 0 + diversity_dropped = 0 if detail == "answer_ready": selection_limit = _LCM_RECALL_LIMIT_CAP if delta_requested else limit expanded_limit = ( @@ -4412,16 +4497,12 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: else _LCM_RECALL_ANSWER_READY_EXPANDED_HIT_LIMIT ) if reference_strict: - ( - selected_entries, - diversity_dropped, - unreferenced_dropped, - ) = _lcm_recall_citable_entries( + strict_selector = _LcmRecallStrictSelector( ordered, - limit=selection_limit, per_session_limit=_LCM_RECALL_ANSWER_READY_PER_SESSION_LIMIT, expanded_limit=expanded_limit, ) + selected_entries = strict_selector.take(selection_limit) else: selected_entries, diversity_dropped = _lcm_recall_diverse_entries( ordered, @@ -4453,7 +4534,20 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: response_chars = 0 response_cap_truncated = False unreferenced_omitted = 0 - for entry in selected_entries: + # Reference-strict shaping validates each admitted candidate against the row + # as it stands NOW: hydration already read the row for the in-budget slots, + # and this one batched read covers the rest, so no delivered span is taken on + # trust from a chunk index that may have gone stale. + strict_rows: dict[int, dict[str, Any]] = {} + if reference_strict: + strict_rows = _lcm_recall_strict_rows( + engine, selected_entries, answer_ready_content + ) + pending = list(selected_entries) + cursor = 0 + while cursor < len(pending): + entry = pending[cursor] + cursor += 1 hit = entry["hit"] arms = sorted({arm_order[index] for index in entry["ranks"].keys()}) item: dict[str, Any] = { @@ -4517,13 +4611,35 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: else "ingest_fallback" ), } - # Fail-closed backstop: the candidate filter admitted this hit on the - # promise of a hydrated window, so only a missing store row can land - # here. Omit rather than deliver an uncitable card; the count is - # disclosed instead of the shortfall being silent. - if reference_strict and not _lcm_recall_item_is_referenced(item): - unreferenced_omitted += 1 - continue + if reference_strict: + # Admission was a prediction; this is the truth check. Publish the + # span the delivered text ACTUALLY occupies so a consumer never has + # to guess an offset, and refill from the ranked tail when a + # candidate fails -- stopping short here would underfill while + # citable candidates sat unexamined just below the cut. + if hydrated is not None: + # The window was cut from the row this same request, so it is + # true by construction and needs no second read. + span = ( + int(hydrated["content_offset"]), + int(hydrated["content_returned_chars"]), + ) + else: + if item.get("content_offset") is None: + item["content_offset"] = hit.get("content_offset") + span = _lcm_recall_verified_span( + item, strict_rows.get(item.get("store_id")) + ) + if span is None: + unreferenced_omitted += 1 + refill = strict_selector.take(1) + if refill: + pending.extend(refill) + strict_rows.update( + _lcm_recall_strict_rows(engine, refill, answer_ready_content) + ) + continue + item["content_offset"], item["content_returned_chars"] = span item_chars = len(json.dumps(item, ensure_ascii=False)) if hits_out and response_chars + item_chars > _LCM_RECALL_RESPONSE_CHAR_CAP: response_cap_truncated = True @@ -4589,13 +4705,20 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: } if reference_strict: expansion["reference_strict"] = True - expansion["unreferenced_dropped_count"] = unreferenced_dropped + expansion["diversity_dropped_count"] = strict_selector.diversity_dropped + expansion["unreferenced_dropped_count"] = ( + strict_selector.unreferenced_dropped + ) expansion["unreferenced_omitted_count"] = unreferenced_omitted expansion["summary_leads"] = summary_leads expansion["reference_policy"] = ( - "no hit lacking a validated (store_id, char_start, char_end) source " - "span is delivered; dropped candidates are backfilled by the " - "next-ranked citable hit, so summaries rank but never cite" + "every delivered hit publishes the (store_id, content_offset, " + "content_returned_chars) span its text occupies in the current " + "row; a candidate that fails that check is replaced by the " + "next-ranked citable one. Summary nodes are never delivered as " + "evidence -- their relevance reaches the ranking through the " + "source messages beneath them, and the nodes themselves come " + "back here as non-evidence drill-down leads" ) response["detail"] = detail response["provenance"]["detail"] = detail From 8ca43fe7625c13bf5dedbdd1220b32cbc5ea651d Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 07:16:56 +0700 Subject: [PATCH 38/54] recall: verify before admitting, not after (delta-2 findings 1, 2, 4) Three findings, one root cause: verification lived AFTER admission as a separate refill pass. Taking the reviewer's structural option -- folding verification into the selection walk removes all three by construction instead of patching each, and deletes more than it adds (_lcm_recall_strict_rows, the refill branch and the post-build re-check are all gone). FINDING 1 -- delta mode bypassed replacement. Delta shaping removed entries before the verification/refill loop ever saw them, so it returned short while valid tail candidates remained (requested 8, returned 7, omission count zero). There is no separate pass left to bypass: the walk only ever yields verified candidates. Delta shaping additionally discards refs the caller has already seen, which is the same underfill, so it now draws from the ranked tail until it reaches limit or the walk is exhausted. FINDING 2 -- failed candidates kept session quota. Quota was charged at admission and never released when verification failed, so five failing same-session candidates exhausted the per-session budget and blocked a valid sixth (0/5 results). Quota is now charged only after a candidate is proven, so a hit that is never delivered cannot spend a slot -- nor is it counted as a diversity drop, because it was not dropped for density. FINDING 4 -- replacement verification was query-per-candidate (57 get_batch calls, 55 singletons, on an 80-candidate repro). The walk prefetches a wave of rows ahead of the cursor in ONE batched read, starting AT the candidate being examined rather than after it. The same repro now costs <= 4 batched reads. The verified snapshot is handed to hydration through the new rows_by_id parameter, so each row is read once per request -- and an admitted candidate can no longer go unhydrated because the row vanished between two separate reads. Regression tests, each mutation-checked to fail without its mechanism: delta draws replacements from the tail; a failed candidate leaves quota untouched; 80 candidates cost <= 4 genuinely batched reads; hydration reuses selection's snapshot. The pure-unit selection tests now supply a stub store, since verification needs a row. F35 gate unchanged: 0/16 fail-closes, 25/25 rendered items, delivered store sets identical to the previous round, whole-suite failure set identical to baseline. --- tests/test_lcm_recall.py | 187 +++++++++++++++++++++++++++- tools.py | 256 +++++++++++++++++++++++++-------------- 2 files changed, 345 insertions(+), 98 deletions(-) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index bd06eeb2d..86f49124b 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -1398,12 +1398,46 @@ def rerank(self, query, documents, *, top_k=None, timeout, model="rerank-2.5-lit # citable one, and the omission count is surfaced rather than silent. -def _message_entry(store_id, *, session_id="session-a", chunk_span=None): +def _stub_row_text(store_id): + return f"row-{store_id} verbatim body text" + + +class _StubStore: + """Minimal store for the selection unit tests; counts batched reads.""" + + def __init__(self, store_ids): + self._rows = { + store_id: {"content": _stub_row_text(store_id)} for store_id in store_ids + } + self.batch_calls = [] + + def get_batch(self, store_ids): + self.batch_calls.append(list(store_ids)) + return {sid: self._rows[sid] for sid in store_ids if sid in self._rows} + + +def _stub_engine(store_ids): + return SimpleNamespace(_store=_StubStore(store_ids)) + + +def _message_entry( + store_id, *, session_id="session-a", chunk_span=None, citable=True, snippet=None +): + """A ranked candidate. + + ``citable=False`` omits the offset a hit needs to be cited without hydration + (the shape a pure-FTS hit has). ``snippet`` overrides the excerpt so a + candidate can be citable-SHAPED yet fail verification against its row -- the + case that forces the walk to keep reading. + """ hit = { "kind": "message_excerpt", "store_id": store_id, "session_id": session_id, } + if citable: + hit["content_offset"] = 0 + hit["snippet"] = _stub_row_text(store_id) if snippet is None else snippet if chunk_span is not None: hit["chunk_span"] = chunk_span return {"hit": hit} @@ -1432,7 +1466,7 @@ def test_reference_strict_backfills_past_every_uncitable_candidate_shape(): _message_entry(2, chunk_span=span), _summary_entry(902, store_id=7), _message_entry(3, chunk_span=span), - _message_entry(4), # no chunk span, and slot 3 is past expanded_limit=2 + _message_entry(4, citable=False), # no offset, and slot 3 is past expanded_limit=2 _message_entry(5, chunk_span=span), _message_entry(6, chunk_span=span), ] @@ -1442,6 +1476,7 @@ def test_reference_strict_backfills_past_every_uncitable_candidate_shape(): limit=5, per_session_limit=5, expanded_limit=2, + engine=_stub_engine(range(1, 10)), ) assert [entry["hit"]["store_id"] for entry in selected] == [1, 2, 3, 5, 6] @@ -1454,13 +1489,14 @@ def test_reference_strict_admits_an_unspanned_message_inside_the_hydration_budge """Slot position decides: a chunk-span-less message is citable while the hydration budget still covers it (it will carry content_offset), and only becomes uncitable once it falls past that budget.""" - ordered = [_message_entry(index) for index in range(1, 5)] + ordered = [_message_entry(index, citable=False) for index in range(1, 5)] selected, _dropped, unreferenced = lcm_tools._lcm_recall_citable_entries( ordered, limit=4, per_session_limit=5, expanded_limit=2, + engine=_stub_engine(range(1, 5)), ) assert [entry["hit"]["store_id"] for entry in selected] == [1, 2] @@ -1478,6 +1514,7 @@ def test_reference_strict_skips_uncitable_before_it_consumes_session_quota(): limit=5, per_session_limit=5, expanded_limit=8, + engine=_stub_engine(range(1, 6)), ) assert [entry["hit"]["store_id"] for entry in selected] == [1, 2, 3, 4, 5] @@ -1671,7 +1708,6 @@ def _forbidden(*_args, **_kwargs): monkeypatch.setattr(lcm_tools, "_LcmRecallStrictSelector", _forbidden) monkeypatch.setattr(lcm_tools, "_lcm_recall_verified_span", _forbidden) - monkeypatch.setattr(lcm_tools, "_lcm_recall_strict_rows", _forbidden) payload = _recall( recall_engine, @@ -1922,3 +1958,146 @@ def rewriting_get_batch(ids): assert delivered <= set(other) for hit in payload["hits"]: assert isinstance(hit["content_offset"], int) + + +# -- Delta-2 review: verification folded INTO the selection walk -------------- +# +# Findings 1, 2 and 4 all came from verification living AFTER admission as a +# separate refill pass: delta shaping could bypass the pass, a failed candidate +# had already spent its session quota, and each replacement paid its own read. +# Verifying before admitting removes all three by construction; finding 3 is the +# independent one -- fusion must keep every citable representation of a row. + + +def test_delta_mode_draws_replacements_from_the_ranked_tail( + recall_engine, monkeypatch +): + """FINDING 1: delta shaping must not silently return short. + + It discards entries whose refs the caller has already seen, which is the + same underfill the resumable walk exists to prevent -- valid tail candidates + were left unexamined while the response came back one hit light. + """ + _only_vector_arms(monkeypatch) + store_ids = _seed_citable_messages( + recall_engine, 30, sessions=tuple(f"session-{i}" for i in range(6)) + ) + + # Delta selects a 25-candidate wave up front, so the caller must have seen + # enough of that wave that satisfying `limit` REQUIRES the ranked tail. + first = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + include="verbatim", + scope_bias=0.0, + limit=25, + ) + assert len(first["hits"]) == 25 + seen = [ + f"lcm:{hit['store_id']}:{hit['content_offset']}-" + f"{hit['content_offset'] + hit['content_returned_chars']}" + for hit in first["hits"][:20] + ] + + delta = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + include="verbatim", + scope_bias=0.0, + limit=8, + seen_refs=seen, + ) + + assert len(delta["hits"]) == 8, "seen refs must be replaced, not just removed" + assert not ({hit["exact_ref"] for hit in delta["hits"]} & set(seen)) + assert {hit["store_id"] for hit in delta["hits"]} <= set(store_ids) + + +def test_failed_candidate_does_not_spend_session_quota(): + """FINDING 2: quota is for DELIVERED hits. + + Charging it at admission let five failing same-session candidates exhaust + the per-session budget and block a valid sixth, returning nothing at all. + """ + per_session = 5 + # Citable-SHAPED but unverifiable, so each one reaches the quota check -- + # a shape-rejected candidate never gets that far and proves nothing here. + ordered = [ + _message_entry(index, session_id="session-a", snippet="not in the row") + for index in range(1, 6) + ] + ordered.append(_message_entry(99, session_id="session-a")) + + selected, dropped, unreferenced = lcm_tools._lcm_recall_citable_entries( + ordered, + limit=5, + per_session_limit=per_session, + expanded_limit=0, + engine=_stub_engine([*range(1, 6), 99]), + ) + + assert [entry["hit"]["store_id"] for entry in selected] == [99] + assert unreferenced == 5 + assert dropped == 0, "a hit that was never delivered cannot be a diversity drop" + + +def test_selection_reads_rows_in_batches_not_one_per_candidate(): + """FINDING 4: the single-batch contract must survive replacement. + + Verifying per replacement made the reader a query-per-candidate path (55 + singleton reads on an 80-candidate repro). + """ + # Citable-SHAPED but unverifiable: each one must be read before it can be + # rejected, which is exactly the path that used to read one row at a time. + ordered = [ + _message_entry(index, session_id=f"session-{index}", snippet="not in the row") + for index in range(1, 80) + ] + ordered.append(_message_entry(99, session_id="session-99")) + engine = _stub_engine([*range(1, 80), 99]) + + selected, _dropped, unreferenced = lcm_tools._lcm_recall_citable_entries( + ordered, + limit=1, + per_session_limit=5, + expanded_limit=0, + engine=engine, + ) + + assert [entry["hit"]["store_id"] for entry in selected] == [99] + assert unreferenced == 79 + calls = engine._store.batch_calls + assert len(calls) <= 4, f"expected batched reads, got {len(calls)} for 80 candidates" + assert max(len(call) for call in calls) > 1, "reads must actually be batched" + + +def test_hydration_reuses_the_rows_selection_already_read( + recall_engine, monkeypatch +): + """The verified snapshot is handed to hydration, so an admitted candidate + cannot go unhydrated because the row vanished between two separate reads.""" + _only_vector_arms(monkeypatch) + _seed_citable_messages(recall_engine, 10) + calls = [] + real_get_batch = recall_engine._store.get_batch + + def counting_get_batch(ids): + calls.append(list(ids)) + return real_get_batch(ids) + + monkeypatch.setattr(recall_engine._store, "get_batch", counting_get_batch) + + payload = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + include="verbatim", + scope_bias=0.0, + limit=8, + ) + + assert len(payload["hits"]) == 8 + # Selection's wave read is the only row read; hydration reuses it. + assert len(calls) == 1, f"expected one batched read, got {len(calls)}" diff --git a/tools.py b/tools.py index afd2c0a20..046be276e 100644 --- a/tools.py +++ b/tools.py @@ -289,6 +289,11 @@ def _parse_strict_int(value: Any, name: str) -> tuple[int | None, str | None]: # the point is to make the SESSION reachable with citable evidence, and the FTS # and chunk arms are what rank individual messages inside it. _LCM_RECALL_SUMMARY_SOURCE_PER_NODE = 4 +# How many candidate rows reference-strict selection reads per batch while it +# walks the ranking. Verification needs the row, so the walk reads AHEAD of the +# cursor in waves: a bounded number of batched reads per request rather than one +# read per candidate it has to skip. +_LCM_RECALL_STRICT_READ_WAVE = 32 _LCM_RECALL_ANSWER_READY_CONTENT_CHARS = 2_400 # Recency boost half-life (30 days) and its floor: a memory's rank_score is # multiplied by 2**(-age/half_life), clamped so age never zeroes an otherwise @@ -3594,94 +3599,143 @@ def _lcm_recall_verified_span( return offset, len(text) -def _lcm_recall_strict_rows( - engine: "LCMEngine", - entries: list[dict[str, Any]], - answer_ready_content: dict[tuple, dict[str, Any]], -) -> dict[int, dict[str, Any]]: - """Read the current rows the un-hydrated admitted candidates claim to quote. - - Hydration already read the row for every in-budget slot and cut the window - out of it, so those spans are true by construction. The candidates past that - budget were admitted on a chunk index alone -- one batched read (never a - query per hit) gives them the same standard of proof. - """ - store_ids = [ - int(entry["hit"]["store_id"]) - for entry in entries - if entry["hit"].get("kind") != "summary" - and entry["hit"].get("store_id") is not None - and _hit_identity(entry["hit"]) not in answer_ready_content - ] - return engine._store.get_batch(store_ids) if store_ids else {} - - def _lcm_recall_citable_entries( ordered: list[dict[str, Any]], *, limit: int, per_session_limit: int, expanded_limit: int, + engine: "LCMEngine" | None = None, ) -> tuple[list[dict[str, Any]], int, int]: - """Reference-strict selection with no reserve (kept for direct unit use).""" + """Reference-strict selection helper (kept for direct unit use).""" selector = _LcmRecallStrictSelector( - ordered, per_session_limit=per_session_limit, expanded_limit=expanded_limit + ordered, + engine=engine, + per_session_limit=per_session_limit, + expanded_limit=expanded_limit, ) selected = selector.take(limit) return selected, selector.diversity_dropped, selector.unreferenced_dropped class _LcmRecallStrictSelector: - """Reference-strict variant of :func:`_lcm_recall_diverse_entries`, resumable. - - Same stable rank-preserving selection with bounded session density, with one - added admission rule: a candidate that cannot plausibly carry a validated - source reference at the slot it would occupy is skipped. Because the rule - filters the RANKED CANDIDATE LIST rather than the finished result, selection - simply continues down the ranking -- an omitted hit is backfilled by the - next-ranked citable one and the delivered count stays at ``limit``. - - The walk is RESUMABLE because admission is only a prediction: a candidate can - still fail verification against its current row once the response is being - shaped (a deleted row, a stale chunk span). Stopping dead at ``limit`` would - then underfill while citable candidates sat unexamined just below the cut, so - the caller draws a replacement with :meth:`take` and the walk picks up where - it left off -- same rules, same session ledger, counters advanced only over - what was actually examined. A run that needs no replacement therefore walks - exactly as far as the non-resumable version did and reports the identical - ``diversity_dropped``; with nothing uncitable at all it is the identity, so - delivery stays byte-identical. - - The reference check runs BEFORE the session-density check so a candidate that - is never delivered cannot consume session quota. + """Rank-ordered admission that VERIFIES a candidate before it is admitted. + + Reference-strict replacement for :func:`_lcm_recall_diverse_entries`. Same + stable rank-preserving walk with bounded session density, plus the rule that + makes the mode meaningful: a candidate is admitted only once its delivered + text has been found at its claimed offset in the CURRENT row. + + Verifying BEFORE admission rather than after is what keeps the invariant + whole. A candidate that fails never becomes a result, so it cannot consume a + session-density slot that a valid lower-ranked candidate needs; the walk just + continues, which IS the backfill -- there is no separate replacement pass for + another code path (delta shaping, say) to bypass. Reads stay batched: the + walk prefetches a wave of rows ahead of the cursor, so a run costs a bounded + number of batched reads rather than one read per replacement. + + Two candidate shapes are proved differently. Inside the hydration budget the + delivered text is a window cut from the row, so the row's existence IS the + proof and the same snapshot is handed to hydration. Past that budget the hit + ships its own excerpt, so the excerpt must be found at its offset. When a row + surfaced through several arms, each arm's representation is tried in turn: + an FTS match window is not a verbatim prefix and will not verify, but the + chunk or summary-source representation of the same row does, and delivering + that is not a swap -- it is the same row, quoted somewhere it really says. """ def __init__( self, ordered: list[dict[str, Any]], *, + engine: "LCMEngine" | None, per_session_limit: int, expanded_limit: int, + wave_size: int = _LCM_RECALL_STRICT_READ_WAVE, ) -> None: self._ordered = ordered + self._engine = engine self._cursor = 0 self._per_session_limit = per_session_limit self._expanded_limit = expanded_limit + self._wave_size = max(1, wave_size) self._session_counts: dict[str, int] = {} self._admitted = 0 + self._examining = 0 + self._prefetched_to = 0 + self.rows: dict[int, dict[str, Any]] = {} + self.batched_reads = 0 self.diversity_dropped = 0 self.unreferenced_dropped = 0 + def exhausted(self) -> bool: + return self._cursor >= len(self._ordered) + + def _prefetch(self) -> bool: + """Read the next wave of candidate rows in ONE batch. + + Starts at the candidate being examined -- NOT after it -- so the row the + walk needs right now is always inside the wave it triggers. + """ + if self._engine is None: + return False + wanted: list[int] = [] + index = max(self._examining, self._prefetched_to) + while index < len(self._ordered) and len(wanted) < self._wave_size: + hit = self._ordered[index]["hit"] + index += 1 + store_id = hit.get("store_id") + if hit.get("kind") == "summary" or store_id is None: + continue + store_id = int(store_id) + if store_id in self.rows or store_id in wanted: + continue + wanted.append(store_id) + if index == self._prefetched_to: + return False + self._prefetched_to = index + if not wanted: + return True + self.batched_reads += 1 + self.rows.update(self._engine._store.get_batch(wanted)) + return True + + def _row_for(self, store_id: Any) -> dict[str, Any] | None: + if store_id is None: + return None + store_id = int(store_id) + while store_id not in self.rows and self._prefetch(): + pass + return self.rows.get(store_id) + + def _verify(self, entry: dict[str, Any], *, hydratable: bool) -> bool: + """Prove this candidate can be cited, adopting a representation if needed.""" + hit = entry["hit"] + row = self._row_for(hit.get("store_id")) + if row is None or not str(row.get("content") or ""): + return False + if hydratable: + # Hydration cuts the delivered window out of this very row. + return True + span = _lcm_recall_verified_span(hit, row) + if span is None: + return False + entry["_strict_span"] = span + return True + def take(self, count: int) -> list[dict[str, Any]]: - """Admit up to ``count`` further candidates, resuming the ranked walk.""" + """Admit up to ``count`` further VERIFIED candidates, resuming the walk.""" taken: list[dict[str, Any]] = [] while self._cursor < len(self._ordered) and len(taken) < count: entry = self._ordered[self._cursor] + self._examining = self._cursor self._cursor += 1 hit = entry["hit"] - if _lcm_recall_reference_shape( - hit, hydratable=self._admitted < self._expanded_limit - ) is None: + hydratable = self._admitted < self._expanded_limit + if _lcm_recall_reference_shape(hit, hydratable=hydratable) is None: + self.unreferenced_dropped += 1 + continue + if not self._verify(entry, hydratable=hydratable): self.unreferenced_dropped += 1 continue raw_session_id = hit.get("session_id") @@ -3735,8 +3789,16 @@ def _lcm_recall_answer_ready_content( *, query: str, expanded_limit: int = _LCM_RECALL_ANSWER_READY_EXPANDED_HIT_LIMIT, + rows_by_id: dict[int, dict[str, Any]] | None = None, ) -> dict[tuple, dict[str, Any]]: - """Hydrate selected exact refs with bounded reads and no retrieval search.""" + """Hydrate selected exact refs with bounded reads and no retrieval search. + + ``rows_by_id`` lets a caller that has ALREADY read these rows hand them over + instead of paying for a second read. Reference-strict selection verifies each + candidate against its row before admitting it, so reusing that same snapshot + also removes the window in which a row could vanish between the two reads and + leave an admitted candidate unhydrated. + """ selected = entries[:expanded_limit] store_ids = [ int(entry["hit"]["store_id"]) @@ -3744,7 +3806,9 @@ def _lcm_recall_answer_ready_content( if entry["hit"].get("kind") == "message_excerpt" and entry["hit"].get("store_id") is not None ] - stored_by_id = engine._store.get_batch(store_ids) + stored_by_id = ( + rows_by_id if rows_by_id is not None else engine._store.get_batch(store_ids) + ) hydrated: dict[tuple, dict[str, Any]] = {} for entry in selected: @@ -4499,33 +4563,63 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: if reference_strict: strict_selector = _LcmRecallStrictSelector( ordered, + engine=engine, per_session_limit=_LCM_RECALL_ANSWER_READY_PER_SESSION_LIMIT, expanded_limit=expanded_limit, ) selected_entries = strict_selector.take(selection_limit) + strict_rows = strict_selector.rows else: selected_entries, diversity_dropped = _lcm_recall_diverse_entries( ordered, limit=selection_limit, per_session_limit=_LCM_RECALL_ANSWER_READY_PER_SESSION_LIMIT, ) + strict_rows = None answer_ready_content = _lcm_recall_answer_ready_content( engine, selected_entries, query=query, expanded_limit=expanded_limit, + rows_by_id=strict_rows, ) if delta_requested: - selected_entries = [ - entry - for entry in selected_entries - if entry["hit"].get("kind") != "summary" - if (exact_ref := _lcm_recall_exact_ref( - entry["hit"], - answer_ready_content.get(_hit_identity(entry["hit"])), - )) is not None - and exact_ref not in seen_refs - ][:limit] + def _novel(entries: list[dict[str, Any]]) -> list[dict[str, Any]]: + return [ + entry + for entry in entries + if entry["hit"].get("kind") != "summary" + if (exact_ref := _lcm_recall_exact_ref( + entry["hit"], + answer_ready_content.get(_hit_identity(entry["hit"])), + )) is not None + and exact_ref not in seen_refs + ] + + selected_entries = _novel(selected_entries) + # Delta shaping discards entries the caller has already seen, so it + # too must be able to draw on the ranked tail -- otherwise the mode + # silently returns short while valid candidates remain, which is the + # very underfill the resumable walk exists to prevent. + while ( + reference_strict + and len(selected_entries) < limit + and not strict_selector.exhausted() + ): + more = strict_selector.take(limit - len(selected_entries)) + if not more: + break + answer_ready_content.update( + _lcm_recall_answer_ready_content( + engine, + more, + query=query, + expanded_limit=len(more), + rows_by_id=strict_selector.rows, + ) + ) + selected_entries.extend(_novel(more)) + selected_entries = selected_entries[:limit] else: selected_entries = ordered diversity_dropped = 0 @@ -4534,20 +4628,7 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: response_chars = 0 response_cap_truncated = False unreferenced_omitted = 0 - # Reference-strict shaping validates each admitted candidate against the row - # as it stands NOW: hydration already read the row for the in-budget slots, - # and this one batched read covers the rest, so no delivered span is taken on - # trust from a chunk index that may have gone stale. - strict_rows: dict[int, dict[str, Any]] = {} - if reference_strict: - strict_rows = _lcm_recall_strict_rows( - engine, selected_entries, answer_ready_content - ) - pending = list(selected_entries) - cursor = 0 - while cursor < len(pending): - entry = pending[cursor] - cursor += 1 + for entry in selected_entries: hit = entry["hit"] arms = sorted({arm_order[index] for index in entry["ranks"].keys()}) item: dict[str, Any] = { @@ -4612,32 +4693,19 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: ), } if reference_strict: - # Admission was a prediction; this is the truth check. Publish the - # span the delivered text ACTUALLY occupies so a consumer never has - # to guess an offset, and refill from the ranked tail when a - # candidate fails -- stopping short here would underfill while - # citable candidates sat unexamined just below the cut. + # Selection already proved this candidate against its row, so there + # is nothing left to re-check here -- only the proven span to + # PUBLISH, so a consumer never has to guess an offset. (__init__.py's + # _answer_ready_baseline substitutes 0 when the field is absent.) if hydrated is not None: - # The window was cut from the row this same request, so it is - # true by construction and needs no second read. span = ( int(hydrated["content_offset"]), int(hydrated["content_returned_chars"]), ) else: - if item.get("content_offset") is None: - item["content_offset"] = hit.get("content_offset") - span = _lcm_recall_verified_span( - item, strict_rows.get(item.get("store_id")) - ) + span = entry.get("_strict_span") if span is None: unreferenced_omitted += 1 - refill = strict_selector.take(1) - if refill: - pending.extend(refill) - strict_rows.update( - _lcm_recall_strict_rows(engine, refill, answer_ready_content) - ) continue item["content_offset"], item["content_returned_chars"] = span item_chars = len(json.dumps(item, ensure_ascii=False)) From 5674cc0d3e9dc293452b3e1e5c7bbf8f982bfe66 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 07:17:10 +0700 Subject: [PATCH 39/54] recall: keep every citable representation of a fused row (delta-2 finding 3) Defect: RRF fuses by identity and keeps ONE base hit per row -- the earliest arm, so FTS whenever it hit. Only the CHUNK arm's metadata was reconciled onto that base afterwards, so any other arm's representation was discarded. An FTS snippet is a match window taken from deep inside the row while the hit claims offset 0, which is not a verbatim span, so past the hydration budget such a row could not be cited at all -- even when the summary-source representation of that same row (its verbatim prefix) was perfectly citable. Repro: nine rows surfaced by both the FTS and summary-source arms and by no chunk hit -- so chunk reconciliation could not help -- and the ninth, the one past the hydration budget, was omitted. Fix: fusion now keeps the union of representations for a row on the entry, and verification tries them in rank order, delivering the first the row actually supports. This is not evidence-swapping (PR #173 finding 3): it is the same row, same identity, same rank, quoted at a place it really says -- the alternative being to drop the row's evidence entirely. Regression test mutation-checked: restricting verification to the fused base alone turns it red. --- tests/test_lcm_recall.py | 48 ++++++++++++++++++++++++++++++++++++++++ tools.py | 46 +++++++++++++++++++++++++++++++++----- 2 files changed, 89 insertions(+), 5 deletions(-) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index 86f49124b..d66a8e15c 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -2043,6 +2043,54 @@ def test_failed_candidate_does_not_spend_session_quota(): assert dropped == 0, "a hit that was never delivered cannot be a diversity drop" +def test_fusion_keeps_a_rows_citable_representation_when_fts_is_the_base( + recall_engine, monkeypatch +): + """FINDING 3: fusion picks ONE base per row; it must not lose the others. + + Nine rows surface through BOTH the FTS arm and the summary-source arm, and + none through the chunk arm -- so the existing chunk reconciliation cannot + help. FTS wins the fused base by arm order, but an FTS snippet is a match + window taken from deep inside the row while the hit claims offset 0, so past + the hydration budget the ninth row was dropped even though the + summary-source representation of that same row (its verbatim prefix) is + perfectly citable. + """ + match = "kanban dashboard sprint" + contents = {} + for index in range(9): + session = f"session-{index}" + store_id = recall_engine._store.append( + session, + {"role": "user", "content": "lead " * 60 + match + f" tail {index}"}, + source="chat", + ) + contents[store_id] = recall_engine._store.get(store_id)["content"] + node = _add_summary( + recall_engine, + f"{match} rollup {index}", + session_id=session, + created_at=10.0, + source_ids=[store_id], + ) + _seed_summary_vectors(recall_engine, [(node, [1.0, 0.0])]) + + payload = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + scope_bias=0.0, + limit=9, + ) + + assert len(payload["hits"]) == 9, "no row may be lost to its FTS representation" + for hit in payload["hits"]: + content = contents[hit["store_id"]] + start = hit["content_offset"] + text = hit.get("content") or hit["snippet"] + assert content[start:start + len(text)] == text + + def test_selection_reads_rows_in_batches_not_one_per_candidate(): """FINDING 4: the single-batch contract must survive replacement. diff --git a/tools.py b/tools.py index 046be276e..d82d46b22 100644 --- a/tools.py +++ b/tools.py @@ -3717,11 +3717,26 @@ def _verify(self, entry: dict[str, Any], *, hydratable: bool) -> bool: if hydratable: # Hydration cuts the delivered window out of this very row. return True - span = _lcm_recall_verified_span(hit, row) - if span is None: - return False - entry["_strict_span"] = span - return True + for candidate in [hit, *entry.get("_alternates", [])]: + probe = { + "kind": hit.get("kind"), + "store_id": hit.get("store_id"), + "snippet": candidate.get("snippet"), + "content_offset": candidate.get("content_offset"), + } + span = _lcm_recall_verified_span(probe, row) + if span is None: + continue + if candidate is not hit: + # Same row, quoted where it really says it. + hit["snippet"] = candidate.get("snippet") + hit["content_offset"] = candidate.get("content_offset") + if candidate.get("chunk_span"): + hit["chunk_span"] = candidate["chunk_span"] + hit["expand_hint"] = _lcm_recall_excerpt_expand_hint(hit) + entry["_strict_span"] = span + return True + return False def take(self, count: int) -> list[dict[str, Any]]: """Admit up to ``count`` further VERIFIED candidates, resuming the walk.""" @@ -4497,12 +4512,33 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: chunk_by_store: dict[Any, dict[str, Any]] = {} for chunk_hit in arm_hits.get("chunk", []): chunk_by_store.setdefault(chunk_hit.get("store_id"), chunk_hit) + # Every OTHER arm's representation of the same row is kept alongside the + # fused base rather than discarded. Fusion picks one base per identity (the + # earliest arm, so FTS when it hit), but an FTS snippet is a match window + # with markers -- not a verbatim span -- so a row whose only surviving + # representation is the FTS one cannot be cited beyond the hydration budget + # even though the chunk or summary-source excerpt of that same row could be. + # Reference-strict verification tries these in order and delivers the first + # that the row actually supports. + alternates_by_store: dict[Any, list[dict[str, Any]]] = {} + for arm_name in ("summary", "chunk"): + for other_hit in arm_hits.get(arm_name, []): + if other_hit.get("kind") != "message_excerpt": + continue + alternates_by_store.setdefault(other_hit.get("store_id"), []).append( + other_hit + ) chunk_arm_index = arm_order.index("chunk") if "chunk" in arm_order else None fts_arm_index = arm_order.index("fts") if "fts" in arm_order else None for entry in ordered: hit = entry["hit"] if hit.get("kind") != "message_excerpt": continue + entry["_alternates"] = [ + other + for other in alternates_by_store.get(hit.get("store_id"), []) + if other is not hit + ] chunk_hit = chunk_by_store.get(hit.get("store_id")) if chunk_hit is None: continue From 5db4e874df4895dcc2166c428b0e51a37e28ddfe Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 07:34:24 +0700 Subject: [PATCH 40/54] recall: a seen delta result must not spend a session slot (delta-3 finding 1) Defect: session quota was charged when the walk admitted a candidate, but delta shaping removes references the caller already holds AFTERWARDS. Those entries kept the density slot they never used, so the replacements behind them failed the per-session check. Repro: 20 citable rows in ONE session, the first five marked seen, delta request for five -> 0 results and 15 diversity drops with 15 novel rows still available. Two halves, both needed: * release(). An entry a later stage discards was never delivered, so it hands its session slot back -- and the hydration budget it had reserved with it. Density bounds what the caller RECEIVES; letting already-seen rows hold slots crowds out the novel ones still waiting in the ranking. * Select a wave at a time. Delta used to select the whole 25-candidate cap up front and filter afterwards, which walked the ranking to EXHAUSTION while the seen entries held quota -- by the time the refund happened there was nothing left to resume into, so the refund alone fixed nothing. Selecting `limit` and refilling lets a released slot be reused by the next wave. With no seen refs this selects exactly the same entries in the same order as before. Regression test mutation-checked against both halves: removing release() or restoring the up-front cap selection each turns it red. --- tests/test_lcm_recall.py | 45 +++++++++++++++++++++++++++ tools.py | 67 +++++++++++++++++++++++++++++++--------- 2 files changed, 97 insertions(+), 15 deletions(-) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index d66a8e15c..4844520fb 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -2149,3 +2149,48 @@ def counting_get_batch(ids): assert len(payload["hits"]) == 8 # Selection's wave read is the only row read; hydration reuses it. assert len(calls) == 1, f"expected one batched read, got {len(calls)}" + + +# -- Delta-3 review ------------------------------------------------------------ + + +def test_seen_delta_results_do_not_spend_session_quota(recall_engine, monkeypatch): + """FINDING 1: an entry the caller already has was never delivered either. + + Quota is charged when the walk admits a candidate, but delta shaping drops + already-seen references afterwards. With 20 citable rows in ONE session and + the first five seen, the session budget was spent on those five and every + replacement then failed the density check -- 0 results and 15 diversity + drops with 15 novel rows still available. + """ + _only_vector_arms(monkeypatch) + store_ids = _seed_citable_messages(recall_engine, 20, sessions=("session-a",)) + + first = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + include="verbatim", + scope_bias=0.0, + limit=5, + ) + assert len(first["hits"]) == 5 + seen = [ + f"lcm:{hit['store_id']}:{hit['content_offset']}-" + f"{hit['content_offset'] + hit['content_returned_chars']}" + for hit in first["hits"] + ] + + delta = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + include="verbatim", + scope_bias=0.0, + limit=5, + seen_refs=seen, + ) + + assert len(delta["hits"]) == 5, "seen entries must refund the slot they never used" + assert not ({hit["exact_ref"] for hit in delta["hits"]} & set(seen)) + assert {hit["store_id"] for hit in delta["hits"]} <= set(store_ids) diff --git a/tools.py b/tools.py index d82d46b22..b8f75ff08 100644 --- a/tools.py +++ b/tools.py @@ -3708,6 +3708,7 @@ def _row_for(self, store_id: Any) -> dict[str, Any] | None: pass return self.rows.get(store_id) + def _verify(self, entry: dict[str, Any], *, hydratable: bool) -> bool: """Prove this candidate can be cited, adopting a representation if needed.""" hit = entry["hit"] @@ -3738,6 +3739,31 @@ def _verify(self, entry: dict[str, Any], *, hydratable: bool) -> bool: return True return False + @staticmethod + def _session_key(hit: dict[str, Any]) -> str: + raw_session_id = hit.get("session_id") + # Missing session identities must not collapse into one synthetic session. + return ( + str(raw_session_id) + if raw_session_id not in {None, ""} + else f"missing:{_hit_identity(hit)!r}" + ) + + def release(self, entry: dict[str, Any]) -> None: + """Give back the slot an admitted candidate turned out not to use. + + Session density bounds what the caller RECEIVES. A candidate that a + later shaping stage discards -- delta dropping a reference the caller + already holds -- was never delivered, so holding its slot would let + already-seen rows crowd out the novel ones still waiting in the ranking. + Releasing also restores the hydration budget it had reserved. + """ + session_key = self._session_key(entry["hit"]) + if self._session_counts.get(session_key): + self._session_counts[session_key] -= 1 + if self._admitted: + self._admitted -= 1 + def take(self, count: int) -> list[dict[str, Any]]: """Admit up to ``count`` further VERIFIED candidates, resuming the walk.""" taken: list[dict[str, Any]] = [] @@ -3753,12 +3779,7 @@ def take(self, count: int) -> list[dict[str, Any]]: if not self._verify(entry, hydratable=hydratable): self.unreferenced_dropped += 1 continue - raw_session_id = hit.get("session_id") - session_key = ( - str(raw_session_id) - if raw_session_id not in {None, ""} - else f"missing:{_hit_identity(hit)!r}" - ) + session_key = self._session_key(hit) if self._session_counts.get(session_key, 0) >= self._per_session_limit: self.diversity_dropped += 1 continue @@ -4590,7 +4611,12 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: # stable post-rank diversity before bounded exact-ref hydration. diversity_dropped = 0 if detail == "answer_ready": - selection_limit = _LCM_RECALL_LIMIT_CAP if delta_requested else limit + # Take only what the response can hold. Delta used to select the whole + # 25-candidate cap up front and filter afterwards, which walked the + # ranking to exhaustion while already-seen entries held session quota -- + # the refill then had nothing left to resume into. Selecting a wave at a + # time lets a released slot be reused by the next wave. + selection_limit = limit expanded_limit = ( _LCM_RECALL_LIMIT_CAP if delta_requested @@ -4621,16 +4647,27 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: ) if delta_requested: def _novel(entries: list[dict[str, Any]]) -> list[dict[str, Any]]: - return [ - entry - for entry in entries - if entry["hit"].get("kind") != "summary" - if (exact_ref := _lcm_recall_exact_ref( + """Keep the entries carrying a reference the caller lacks. + + A discarded entry hands its session slot back: it is not part of + the response, so it must not count against the density budget + that decides which novel rows can still be delivered. + """ + kept: list[dict[str, Any]] = [] + for entry in entries: + exact_ref = ( + None + if entry["hit"].get("kind") == "summary" + else _lcm_recall_exact_ref( entry["hit"], answer_ready_content.get(_hit_identity(entry["hit"])), - )) is not None - and exact_ref not in seen_refs - ] + ) + ) + if exact_ref is not None and exact_ref not in seen_refs: + kept.append(entry) + elif reference_strict: + strict_selector.release(entry) + return kept selected_entries = _novel(selected_entries) # Delta shaping discards entries the caller has already seen, so it From d9ab8f96e8fb157adf351a727ab53fbdca6e6f7e Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 07:34:38 +0700 Subject: [PATCH 41/54] recall: a missing row must not cascade waves or pin the corpus (delta-3 finding 2) Defect: the walk tracked only which rows it HAD, so it could not tell "not prefetched yet" from "prefetched and absent". Looking for a row the store does not have, it kept calling for another wave -- loading every remaining wave and retaining all of it -- to reject a single candidate. Repro: 100 candidates, row 1 missing, limit=1, wave_size=32 -> four reads [32, 32, 32, 4] and 99 rows retained to select row 2. Fix, both halves the review named: * Remember what was ATTEMPTED, not what came back. A store_id included in a prefetch batch is settled either way, so a miss advances the cursor instead of triggering another wave. The repro now costs ONE read. * Release rows outside the active wave. Only rows a delivered hit depends on need to survive -- hydration reads from this same snapshot -- so a candidate rejected for shape, verification or density drops its row. Retention is bounded by the wave plus what was admitted rather than growing with the corpus: rejecting 99 candidates now retains <= 2 rows instead of 100. Regression tests mutation-checked separately, because the first one alone does not cover the second: with the attempted-set fix in place the missing-row repro reads a single wave, so retention stays bounded whether or not eviction works. The added test rejects 99 candidates that each require a real read. --- tests/test_lcm_recall.py | 59 ++++++++++++++++++++++++++++++++++++++++ tools.py | 23 ++++++++++++++-- 2 files changed, 80 insertions(+), 2 deletions(-) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index 4844520fb..bceda4019 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -2194,3 +2194,62 @@ def test_seen_delta_results_do_not_spend_session_quota(recall_engine, monkeypatc assert len(delta["hits"]) == 5, "seen entries must refund the slot they never used" assert not ({hit["exact_ref"] for hit in delta["hits"]} & set(seen)) assert {hit["store_id"] for hit in delta["hits"]} <= set(store_ids) + + +def test_a_missing_row_does_not_cascade_reads_or_retain_the_corpus(): + """FINDING 2: 'not prefetched' and 'prefetched but absent' are different. + + Without that distinction the walk kept calling for more waves looking for a + row that will never arrive, loading every remaining wave and retaining all + of it: 100 candidates with row 1 missing cost four reads [32,32,32,4] and + held 99 rows just to select row 2. + """ + ordered = [ + _message_entry(index, session_id=f"session-{index}") + for index in range(1, 101) + ] + engine = _stub_engine(range(2, 101)) # row 1 is gone + + selector = lcm_tools._LcmRecallStrictSelector( + ordered, + engine=engine, + per_session_limit=5, + expanded_limit=0, + wave_size=32, + ) + selected = selector.take(1) + + assert [entry["hit"]["store_id"] for entry in selected] == [2] + calls = engine._store.batch_calls + assert len(calls) == 1, f"a missing row must not cascade waves: {[len(c) for c in calls]}" + assert len(selector.rows) <= 32, f"retained {len(selector.rows)} rows, expected one wave" + + +def test_rejected_candidate_rows_are_not_retained(): + """FINDING 2 (retention half): rows outside the active wave are released. + + Reading in waves bounds the reads, but holding on to every row the walk + rejected would still grow with the corpus. Only rows a delivered hit + depends on -- hydration reads from this same snapshot -- need to survive. + """ + ordered = [ + _message_entry(index, session_id=f"session-{index}", snippet="not in the row") + for index in range(1, 100) + ] + ordered.append(_message_entry(100, session_id="session-100")) + engine = _stub_engine(range(1, 101)) + + selector = lcm_tools._LcmRecallStrictSelector( + ordered, + engine=engine, + per_session_limit=5, + expanded_limit=0, + wave_size=32, + ) + selected = selector.take(1) + + assert [entry["hit"]["store_id"] for entry in selected] == [100] + assert selector.unreferenced_dropped == 99 + assert len(selector.rows) <= 2, ( + f"retained {len(selector.rows)} rows after rejecting 99 candidates" + ) diff --git a/tools.py b/tools.py index b8f75ff08..bdff76f99 100644 --- a/tools.py +++ b/tools.py @@ -3663,6 +3663,8 @@ def __init__( self._admitted = 0 self._examining = 0 self._prefetched_to = 0 + self._attempted: set[int] = set() + self._admitted_stores: set[int] = set() self.rows: dict[int, dict[str, Any]] = {} self.batched_reads = 0 self.diversity_dropped = 0 @@ -3688,7 +3690,7 @@ def _prefetch(self) -> bool: if hit.get("kind") == "summary" or store_id is None: continue store_id = int(store_id) - if store_id in self.rows or store_id in wanted: + if store_id in self._attempted or store_id in wanted: continue wanted.append(store_id) if index == self._prefetched_to: @@ -3697,6 +3699,11 @@ def _prefetch(self) -> bool: if not wanted: return True self.batched_reads += 1 + # ATTEMPTED, not merely returned: a row the store does not have must be + # remembered as looked-for, or the walk cannot tell "not fetched yet" + # from "fetched and absent" and keeps calling for waves that can never + # contain it -- pulling in the rest of the corpus to reject one row. + self._attempted.update(wanted) self.rows.update(self._engine._store.get_batch(wanted)) return True @@ -3704,10 +3711,17 @@ def _row_for(self, store_id: Any) -> dict[str, Any] | None: if store_id is None: return None store_id = int(store_id) - while store_id not in self.rows and self._prefetch(): + while store_id not in self._attempted and self._prefetch(): pass return self.rows.get(store_id) + def _forget(self, store_id: Any) -> None: + """Drop a row no delivered hit depends on, bounding retention.""" + if store_id is None: + return + store_id = int(store_id) + if store_id not in self._admitted_stores: + self.rows.pop(store_id, None) def _verify(self, entry: dict[str, Any], *, hydratable: bool) -> bool: """Prove this candidate can be cited, adopting a representation if needed.""" @@ -3775,18 +3789,23 @@ def take(self, count: int) -> list[dict[str, Any]]: hydratable = self._admitted < self._expanded_limit if _lcm_recall_reference_shape(hit, hydratable=hydratable) is None: self.unreferenced_dropped += 1 + self._forget(hit.get("store_id")) continue if not self._verify(entry, hydratable=hydratable): self.unreferenced_dropped += 1 + self._forget(hit.get("store_id")) continue session_key = self._session_key(hit) if self._session_counts.get(session_key, 0) >= self._per_session_limit: self.diversity_dropped += 1 + self._forget(hit.get("store_id")) continue self._session_counts[session_key] = ( self._session_counts.get(session_key, 0) + 1 ) self._admitted += 1 + if hit.get("store_id") is not None: + self._admitted_stores.add(int(hit["store_id"])) taken.append(entry) return taken From 1a0b0e25543441fc7d022735cf862f322f97a571 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 08:00:49 +0700 Subject: [PATCH 42/54] recall: make selection accounting a ledger, not counters (delta-4 findings 1, 2) One commit, because this is one state machine: quota, hydration budget and row retention are three views of a single fact -- whether an entry is still live -- and landing them separately would ship a half-wired ledger with the same class of seam the last three rounds each found. The diagnosis was right. Session density, the hydration budget and row retention were counters maintained by hand at four call sites, and each round found a different one leaking: a refund that never happened (round 3), a refund that arrived after the walk had already consumed the ranking, a refund applied twice, a store that stayed pinned after release. Fixing them one at a time was always going to keep finding the next seam. The ledger: PENDING -> ADMITTED -> (DELIVERED | RELEASED), transitions one-way. Resources are charged on ADMITTED and refunded exactly once on RELEASED, in one place. Retention is DERIVED -- a store is held iff a live entry holds it -- so it cannot drift from the lifecycle. A second release is a counted no-op instead of a second refund. FINDING 1 -- selection consumed the ranking before refunds could land. Asking for more than one session can supply (limit 10, cap 5) made the first wave walk to EXHAUSTION, density-dropping 15 candidates, so releasing the five seen entries had nothing left to resume into. The missing distinction: failing the shape test or verification is a PERMANENT property of a candidate, but being over the session cap is not -- density is refundable. Density-blocked candidates are now DEFERRED in rank order and reconsidered on the next wave, bounded by a reserve (no more slots can be refunded than were admitted). take() also became a TARGET rather than a count, so a wave asks for what the response is still missing and a freed slot is exactly what it refills. FINDING 2 -- release() decremented aggregates with no per-entry ownership, so releasing one entry twice returned two slots (three deliveries against a cap of two) and released stores stayed in the admitted-store set, defeating eviction (100 admitted-then-released entries retained all 100 rows). Both are gone: the transition guard makes the refund per-entry and idempotent, and retention reads the ledger instead of a parallel set. Tests, all mutation-checked: the limit-over-cap refund repro; double release as a counted no-op; a released entry releasing its row; take() topping up to a target; and a seeded 200-step randomized admit/deliver/release/double-release sequence asserting after EVERY step that per-session live counts stay under the cap, the books agree with the per-entry records, released entries never revert, retention stays within wave + live + reserve, and the final delivered set equals the ledger's DELIVERED set. Gate population unchanged: F35 replay 0/16 fail-closes, 25/25 rendered items, 400/400 hits publishing their offset, store-set churn +0/-0. Whole-suite failure set identical to baseline. --- tests/test_lcm_recall.py | 207 ++++++++++++++++++++++++++++++++ tools.py | 246 +++++++++++++++++++++++++++++++-------- 2 files changed, 406 insertions(+), 47 deletions(-) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index bceda4019..995175bd0 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -2253,3 +2253,210 @@ def test_rejected_candidate_rows_are_not_retained(): assert len(selector.rows) <= 2, ( f"retained {len(selector.rows)} rows after rejecting 99 candidates" ) + + +# -- Delta-4 review: the selection ledger -------------------------------------- + + +def test_refund_reaches_the_ranking_when_limit_exceeds_the_session_cap( + recall_engine, monkeypatch +): + """FINDING 1: selection must not consume the ranking before refunds land. + + Asking for more than one session can supply made the first wave walk to + EXHAUSTION -- admitting 5, density-dropping the other 15 -- so when delta + shaping released the five seen entries there was nothing left to resume + into. A density block is refundable, unlike a shape or verification + failure, so those candidates are held in rank order instead of consumed. + """ + _only_vector_arms(monkeypatch) + store_ids = _seed_citable_messages(recall_engine, 20, sessions=("session-a",)) + + first = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + include="verbatim", + scope_bias=0.0, + limit=10, + ) + # One session, cap 5 -- the response can never exceed the cap. + assert len(first["hits"]) == 5 + seen = [ + f"lcm:{hit['store_id']}:{hit['content_offset']}-" + f"{hit['content_offset'] + hit['content_returned_chars']}" + for hit in first["hits"] + ] + + delta = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + include="verbatim", + scope_bias=0.0, + limit=10, + seen_refs=seen, + ) + + assert len(delta["hits"]) == 5, "the refund must reach the deferred candidates" + assert not ({hit["exact_ref"] for hit in delta["hits"]} & set(seen)) + assert {hit["store_id"] for hit in delta["hits"]} <= set(store_ids) + + +def test_double_release_is_a_counted_no_op_not_a_second_refund(): + """FINDING 2: refunds are per-entry, not aggregate decrements. + + Releasing one entry twice used to give the session two slots back, letting + three hits through a cap of two. + """ + ordered = [ + _message_entry(index, session_id="session-a") for index in range(1, 6) + ] + engine = _stub_engine(range(1, 6)) + selector = lcm_tools._LcmRecallStrictSelector( + ordered, engine=engine, per_session_limit=2, expanded_limit=0 + ) + + taken = selector.take(2) + assert len(taken) == 2 + + assert selector.release(taken[0]) is True + assert selector.release(taken[0]) is False, "second release must be a no-op" + assert selector.release(taken[0]) is False + assert selector.ledger.double_releases == 2 + assert selector.ledger.live_count == 1 + assert selector.ledger.session_count("session-a") == 1 + + # Exactly ONE slot came back, so the cap of two still holds. + selector.take(2) + assert selector.ledger.session_count("session-a") == 2 + + +def test_released_entry_stops_pinning_its_row(): + """FINDING 2: retention is derived from the ledger, not a parallel set. + + Released stores stayed in the admitted-store set, so eviction never fired + and 100 admitted-then-released entries retained all 100 rows. + """ + ordered = [ + _message_entry(index, session_id=f"session-{index}") + for index in range(1, 101) + ] + engine = _stub_engine(range(1, 101)) + selector = lcm_tools._LcmRecallStrictSelector( + ordered, engine=engine, per_session_limit=5, expanded_limit=0, wave_size=32 + ) + + for _ in range(100): + taken = selector.take(selector.ledger.live_count + 1) + if not taken: + break + selector.release(taken[0]) + + assert selector.ledger.live_count == 0 + assert len(selector.rows) <= 32, ( + f"retained {len(selector.rows)} rows with nothing live" + ) + + +def test_ledger_invariants_hold_under_a_randomized_transition_sequence(): + """Property check: no admit/deliver/release interleaving can break the books. + + Three rounds of review each found a different counter leaking at a different + seam, so the accounting is exercised as a state machine rather than only at + the seams already known to have failed. + """ + import random + + rng = random.Random(20260729) + per_session_limit = 3 + wave_size = 16 + reserve = 25 + sessions = [f"session-{index % 7}" for index in range(1, 121)] + ordered = [ + _message_entry(index, session_id=sessions[index - 1]) + for index in range(1, 121) + ] + engine = _stub_engine(range(1, 121)) + selector = lcm_tools._LcmRecallStrictSelector( + ordered, + engine=engine, + per_session_limit=per_session_limit, + expanded_limit=0, + wave_size=wave_size, + defer_reserve=reserve, + ) + ledger = selector.ledger + admitted: list[dict] = [] + delivered: list[dict] = [] + released: list[dict] = [] + + def check(step): + # per-session live count never exceeds the cap + for key in {selector._session_key(e["hit"]) for e in ordered}: + assert ledger.session_count(key) <= per_session_limit, f"step {step}: {key}" + # refunds are never double-applied: the books agree with the records + live = [e for e in admitted if ledger.state(e) in ("admitted", "delivered")] + assert ledger.live_count == len(live), f"step {step}: live drift" + assert sum(ledger._session_counts.values()) == ledger.live_count, ( + f"step {step}: session totals drift" + ) + # a released entry never returns to a live state + for entry in released: + assert ledger.state(entry) == "released", f"step {step}: state reversed" + # retention stays bounded by the wave plus what is still in play + assert len(selector.rows) <= wave_size + ledger.live_count + reserve, ( + f"step {step}: retained {len(selector.rows)}" + ) + + for step in range(200): + action = rng.random() + if action < 0.4: + for entry in selector.take(ledger.live_count + rng.randint(1, 4)): + admitted.append(entry) + elif action < 0.6 and admitted: + entry = rng.choice(admitted) + if ledger.deliver(entry): + delivered.append(entry) + elif action < 0.9 and admitted: + entry = rng.choice(admitted) + if selector.release(entry): + released.append(entry) + elif released: + before = ledger.live_count + assert selector.release(rng.choice(released)) is False + assert ledger.live_count == before, f"step {step}: double release refunded" + check(step) + + assert delivered, "the sequence must actually deliver something" + assert set(map(id, ledger.delivered_entries())) == set(map(id, delivered)) + assert ledger.double_releases > 0, "the sequence must exercise double release" + + +def test_take_tops_up_to_a_target_rather_than_adding_a_count(): + """A wave asks for what the response is still MISSING. + + Expressed as a count, a wave after a partial refund would admit a full + count on top of what is already live and overshoot the response; expressed + as a target, a refunded slot is exactly what the next wave refills. + """ + ordered = [ + _message_entry(index, session_id=f"session-{index}") + for index in range(1, 21) + ] + selector = lcm_tools._LcmRecallStrictSelector( + ordered, + engine=_stub_engine(range(1, 21)), + per_session_limit=5, + expanded_limit=0, + ) + + taken = selector.take(5) + assert len(taken) == 5 and selector.ledger.live_count == 5 + assert selector.take(5) == [], "already at the target -- nothing to add" + assert selector.ledger.live_count == 5 + + assert selector.release(taken[0]) is True + assert selector.ledger.live_count == 4 + assert len(selector.take(5)) == 1, "a wave refills exactly the freed slot" + assert selector.ledger.live_count == 5 diff --git a/tools.py b/tools.py index bdff76f99..f6308b432 100644 --- a/tools.py +++ b/tools.py @@ -294,6 +294,10 @@ def _parse_strict_int(value: Any, name: str) -> tuple[int | None, str | None]: # cursor in waves: a bounded number of batched reads per request rather than one # read per candidate it has to skip. _LCM_RECALL_STRICT_READ_WAVE = 32 +# How many density-blocked candidates the walk keeps in rank order awaiting a +# possible refund. Bounded because no more slots can ever be handed back than +# were admitted, so a reserve the size of the response cap is always enough. +_LCM_RECALL_STRICT_DEFER_RESERVE = _LCM_RECALL_LIMIT_CAP _LCM_RECALL_ANSWER_READY_CONTENT_CHARS = 2_400 # Recency boost half-life (30 days) and its floor: a memory's rank_score is # multiplied by 2**(-age/half_life), clamped so age never zeroes an otherwise @@ -3618,6 +3622,100 @@ def _lcm_recall_citable_entries( return selected, selector.diversity_dropped, selector.unreferenced_dropped +class _LcmRecallSelectionLedger: + """Per-entry lifecycle, and every resource an entry holds while it is live. + + Session density, the hydration budget and row retention used to be three + counters maintained by hand at four call sites, and each review round found a + different seam where one of them leaked: a refund that never happened, a + refund applied twice, a store that stayed pinned after its entry was let go. + All three are consequences of ONE fact -- whether an entry is still live -- + so they are DERIVED from that fact here instead of tracked alongside it. + + ``PENDING -> ADMITTED -> (DELIVERED | RELEASED)``. Transitions are one-way. + Resources are charged on ADMITTED and refunded exactly once on RELEASED; + DELIVERED is terminal and keeps them. A second release is a counted no-op, + not a second refund -- the double-refund that let three hits through a cap + of two. + """ + + ADMITTED = "admitted" + DELIVERED = "delivered" + RELEASED = "released" + + def __init__(self) -> None: + # Keyed by id(); the entry itself is held so the id cannot be recycled. + self._records: dict[int, dict[str, Any]] = {} + self._session_counts: dict[str, int] = {} + self._live_stores: dict[int, int] = {} + self._live = 0 + self.double_releases = 0 + + def state(self, entry: dict[str, Any]) -> str | None: + record = self._records.get(id(entry)) + return record["state"] if record else None + + @property + def live_count(self) -> int: + """ADMITTED or DELIVERED -- the entries currently holding resources.""" + return self._live + + def delivered_entries(self) -> list[dict[str, Any]]: + return [ + record["entry"] + for record in self._records.values() + if record["state"] == self.DELIVERED + ] + + def session_count(self, session_key: str) -> int: + return self._session_counts.get(session_key, 0) + + def holds_store(self, store_id: Any) -> bool: + return store_id is not None and int(store_id) in self._live_stores + + def admit( + self, entry: dict[str, Any], *, session_key: str, store_id: Any + ) -> None: + """PENDING -> ADMITTED, charging every resource in this one place.""" + self._records[id(entry)] = { + "entry": entry, + "state": self.ADMITTED, + "session_key": session_key, + "store_id": None if store_id is None else int(store_id), + } + self._session_counts[session_key] = self._session_counts.get(session_key, 0) + 1 + if store_id is not None: + key = int(store_id) + self._live_stores[key] = self._live_stores.get(key, 0) + 1 + self._live += 1 + + def deliver(self, entry: dict[str, Any]) -> bool: + """ADMITTED -> DELIVERED. Terminal; the entry keeps what it holds.""" + record = self._records.get(id(entry)) + if record is None or record["state"] != self.ADMITTED: + return False + record["state"] = self.DELIVERED + return True + + def release(self, entry: dict[str, Any]) -> bool: + """ADMITTED -> RELEASED, refunding once. Idempotent by construction.""" + record = self._records.get(id(entry)) + if record is None or record["state"] != self.ADMITTED: + self.double_releases += 1 + return False + record["state"] = self.RELEASED + session_key = record["session_key"] + if self._session_counts.get(session_key): + self._session_counts[session_key] -= 1 + store_id = record["store_id"] + if store_id is not None and self._live_stores.get(store_id): + self._live_stores[store_id] -= 1 + if not self._live_stores[store_id]: + del self._live_stores[store_id] + self._live -= 1 + return True + + class _LcmRecallStrictSelector: """Rank-ordered admission that VERIFIES a candidate before it is admitted. @@ -3626,22 +3724,32 @@ class _LcmRecallStrictSelector: makes the mode meaningful: a candidate is admitted only once its delivered text has been found at its claimed offset in the CURRENT row. - Verifying BEFORE admission rather than after is what keeps the invariant - whole. A candidate that fails never becomes a result, so it cannot consume a - session-density slot that a valid lower-ranked candidate needs; the walk just - continues, which IS the backfill -- there is no separate replacement pass for - another code path (delta shaping, say) to bypass. Reads stay batched: the - walk prefetches a wave of rows ahead of the cursor, so a run costs a bounded - number of batched reads rather than one read per replacement. + Verifying BEFORE admission is what keeps the invariant whole. A candidate + that fails never becomes a result, so it cannot spend a session slot a valid + lower-ranked candidate needs; the walk simply continues, which IS the + backfill -- there is no separate replacement pass for another stage to + bypass. Reads stay batched: the walk prefetches a wave of rows ahead of the + cursor, and remembers what it ATTEMPTED, so a row the store does not have + settles the candidate instead of dragging in the rest of the corpus. + + Two rejections, two different lifetimes. Failing the shape test or + verification is a PERMANENT property of the candidate, so it is discarded and + its row released. Being over the session cap is not -- density is a + refundable resource, and a later stage handing back a slot (delta dropping a + reference the caller already holds) can make a blocked candidate admissible. + Those are DEFERRED in rank order and reconsidered on the next wave, which is + what lets a refund actually reach the ranking instead of arriving after the + walk has consumed it. The reserve is bounded, because no more slots can ever + be refunded than were admitted. Two candidate shapes are proved differently. Inside the hydration budget the delivered text is a window cut from the row, so the row's existence IS the proof and the same snapshot is handed to hydration. Past that budget the hit ships its own excerpt, so the excerpt must be found at its offset. When a row - surfaced through several arms, each arm's representation is tried in turn: - an FTS match window is not a verbatim prefix and will not verify, but the - chunk or summary-source representation of the same row does, and delivering - that is not a swap -- it is the same row, quoted somewhere it really says. + surfaced through several arms, each arm's representation is tried in turn: an + FTS match window is not a verbatim prefix and will not verify, but the chunk + or summary-source representation of the same row does, and delivering that is + not a swap -- it is the same row, quoted somewhere it really says. """ def __init__( @@ -3652,6 +3760,7 @@ def __init__( per_session_limit: int, expanded_limit: int, wave_size: int = _LCM_RECALL_STRICT_READ_WAVE, + defer_reserve: int = _LCM_RECALL_STRICT_DEFER_RESERVE, ) -> None: self._ordered = ordered self._engine = engine @@ -3659,19 +3768,34 @@ def __init__( self._per_session_limit = per_session_limit self._expanded_limit = expanded_limit self._wave_size = max(1, wave_size) - self._session_counts: dict[str, int] = {} - self._admitted = 0 + self._defer_reserve = max(0, defer_reserve) + self._deferred: list[dict[str, Any]] = [] + self._density_dropped = 0 self._examining = 0 self._prefetched_to = 0 self._attempted: set[int] = set() - self._admitted_stores: set[int] = set() + self.ledger = _LcmRecallSelectionLedger() self.rows: dict[int, dict[str, Any]] = {} self.batched_reads = 0 - self.diversity_dropped = 0 self.unreferenced_dropped = 0 + @property + def diversity_dropped(self) -> int: + """Candidates the density cap kept out, including those still deferred.""" + return self._density_dropped + len(self._deferred) + def exhausted(self) -> bool: - return self._cursor >= len(self._ordered) + return self._cursor >= len(self._ordered) and not self._deferred + + def deliver(self, entry: dict[str, Any]) -> bool: + return self.ledger.deliver(entry) + + def release(self, entry: dict[str, Any]) -> bool: + """Hand back the slot an entry a later stage discarded never used.""" + released = self.ledger.release(entry) + if released: + self._forget(entry["hit"].get("store_id")) + return released def _prefetch(self) -> bool: """Read the next wave of candidate rows in ONE batch. @@ -3716,12 +3840,17 @@ def _row_for(self, store_id: Any) -> dict[str, Any] | None: return self.rows.get(store_id) def _forget(self, store_id: Any) -> None: - """Drop a row no delivered hit depends on, bounding retention.""" + """Drop a row nothing live or still-in-play depends on.""" if store_id is None: return store_id = int(store_id) - if store_id not in self._admitted_stores: - self.rows.pop(store_id, None) + if self.ledger.holds_store(store_id): + return + if any( + entry["hit"].get("store_id") == store_id for entry in self._deferred + ): + return + self.rows.pop(store_id, None) def _verify(self, entry: dict[str, Any], *, hydratable: bool) -> bool: """Prove this candidate can be cited, adopting a representation if needed.""" @@ -3763,30 +3892,34 @@ def _session_key(hit: dict[str, Any]) -> str: else f"missing:{_hit_identity(hit)!r}" ) - def release(self, entry: dict[str, Any]) -> None: - """Give back the slot an admitted candidate turned out not to use. + def _admit(self, entry: dict[str, Any]) -> None: + hit = entry["hit"] + self.ledger.admit( + entry, + session_key=self._session_key(hit), + store_id=hit.get("store_id"), + ) - Session density bounds what the caller RECEIVES. A candidate that a - later shaping stage discards -- delta dropping a reference the caller - already holds -- was never delivered, so holding its slot would let - already-seen rows crowd out the novel ones still waiting in the ranking. - Releasing also restores the hydration budget it had reserved. - """ - session_key = self._session_key(entry["hit"]) - if self._session_counts.get(session_key): - self._session_counts[session_key] -= 1 - if self._admitted: - self._admitted -= 1 + def _take_deferred(self) -> dict[str, Any] | None: + """Reconsider density-blocked candidates a refund may have unblocked.""" + for index, entry in enumerate(self._deferred): + session_key = self._session_key(entry["hit"]) + if self.ledger.session_count(session_key) < self._per_session_limit: + del self._deferred[index] + self._admit(entry) + return entry + return None - def take(self, count: int) -> list[dict[str, Any]]: - """Admit up to ``count`` further VERIFIED candidates, resuming the walk.""" - taken: list[dict[str, Any]] = [] - while self._cursor < len(self._ordered) and len(taken) < count: + def _next_admissible(self) -> dict[str, Any] | None: + admitted = self._take_deferred() + if admitted is not None: + return admitted + while self._cursor < len(self._ordered): entry = self._ordered[self._cursor] self._examining = self._cursor self._cursor += 1 hit = entry["hit"] - hydratable = self._admitted < self._expanded_limit + hydratable = self.ledger.live_count < self._expanded_limit if _lcm_recall_reference_shape(hit, hydratable=hydratable) is None: self.unreferenced_dropped += 1 self._forget(hit.get("store_id")) @@ -3796,20 +3929,37 @@ def take(self, count: int) -> list[dict[str, Any]]: self._forget(hit.get("store_id")) continue session_key = self._session_key(hit) - if self._session_counts.get(session_key, 0) >= self._per_session_limit: - self.diversity_dropped += 1 - self._forget(hit.get("store_id")) + if self.ledger.session_count(session_key) >= self._per_session_limit: + # Refundable, so hold it in rank order rather than discarding it + # -- but only up to the reserve, since no more slots can be + # refunded than were admitted. + if len(self._deferred) < self._defer_reserve: + self._deferred.append(entry) + else: + self._density_dropped += 1 + self._forget(hit.get("store_id")) continue - self._session_counts[session_key] = ( - self._session_counts.get(session_key, 0) + 1 - ) - self._admitted += 1 - if hit.get("store_id") is not None: - self._admitted_stores.add(int(hit["store_id"])) + self._admit(entry) + return entry + return self._take_deferred() + + def take(self, target: int) -> list[dict[str, Any]]: + """Top the live set up to ``target`` verified candidates. + + Expressed as a target rather than a count so a slot handed back between + waves is immediately reusable: the next wave asks for whatever the + response is still missing. + """ + taken: list[dict[str, Any]] = [] + while self.ledger.live_count < target: + entry = self._next_admissible() + if entry is None: + break taken.append(entry) return taken + def _lcm_recall_content_window( content: Any, *, @@ -4698,7 +4848,7 @@ def _novel(entries: list[dict[str, Any]]) -> list[dict[str, Any]]: and len(selected_entries) < limit and not strict_selector.exhausted() ): - more = strict_selector.take(limit - len(selected_entries)) + more = strict_selector.take(limit) if not more: break answer_ready_content.update( @@ -4805,6 +4955,8 @@ def _novel(entries: list[dict[str, Any]]) -> list[dict[str, Any]]: response_cap_truncated = True break response_chars += item_chars + if reference_strict: + strict_selector.deliver(entry) hits_out.append(item) if len(hits_out) >= limit: break From 92d6fcd0009a2b7fe5045948e6035b541a06a2ec Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 08:25:59 +0700 Subject: [PATCH 43/54] recall: make the defer reserve a position, not a buffer (delta-5) Defect: the reserve was sized on the reasoning that no more slots can be refunded than were admitted. That holds for SIMULTANEOUS admissions, but delta processing admits and releases in SEQUENCE, so a long run of already-seen references refunds far more slots than were ever live at once. Once the 25-entry buffer filled, every later density-blocked candidate was discarded outright -- permanently, while still ranked and perfectly citable. Repro: 100 ranked rows in one session, cap 5, the first 30 references seen. The reserve filled with rows 6-30, rows 31-100 were dropped, and paging past the seen run returned EMPTY even though rows 31-35 were novel and valid. Fix: take the review's suggestion -- the candidates are already rank-ordered, so a blocked one is rediscoverable by POSITION and nothing needs to be stored. The walk remembers the position of the earliest density-blocked candidate and rewinds there when a refund frees a slot. With no buffer there is no bound, and with no bound nothing can be discarded. Three supporting changes fall out of that: * Blocked candidates release their rows. A position costs nothing to keep, so retention now tracks only what is live -- strictly tighter than before, where the reserve pinned up to 25 extra rows. * An evicted row must be re-fetchable, so "attempted" splits into rows we HOLD and rows the store does not have. The delta-3 property is unchanged -- a missing row still settles its candidate and never cascades waves -- but a row merely evicted can be read again, in a wave, when its candidate comes back. * Verification is remembered per candidate so a rewind costs a scan rather than a second round of reads, and remembered against the budget it was proved under, since an in-budget candidate is proved only by its row existing -- a weaker claim than the excerpt check a post-budget slot needs. Tests: the reviewer's 100-row/cap-5/30-seen repro, plus the property test's retention bound tightened from wave + live + reserve to wave + live. Four mutations checked, all caught: reinstating a bounded buffer, dropping the rewind, blocking re-fetch of evicted rows, and letting blocked candidates keep their rows. The property test independently caught two of them. Gate population unchanged: F35 replay 0/16 fail-closes, 25/25 rendered items, 400/400 hits publishing their offset, store-set churn +0/-0. Whole-suite failure set identical to baseline. --- tests/test_lcm_recall.py | 60 ++++++++++++++++++-- tools.py | 117 +++++++++++++++++++++++---------------- 2 files changed, 124 insertions(+), 53 deletions(-) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index 995175bd0..553778e5e 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -2371,7 +2371,6 @@ def test_ledger_invariants_hold_under_a_randomized_transition_sequence(): rng = random.Random(20260729) per_session_limit = 3 wave_size = 16 - reserve = 25 sessions = [f"session-{index % 7}" for index in range(1, 121)] ordered = [ _message_entry(index, session_id=sessions[index - 1]) @@ -2384,7 +2383,6 @@ def test_ledger_invariants_hold_under_a_randomized_transition_sequence(): per_session_limit=per_session_limit, expanded_limit=0, wave_size=wave_size, - defer_reserve=reserve, ) ledger = selector.ledger admitted: list[dict] = [] @@ -2404,8 +2402,9 @@ def check(step): # a released entry never returns to a live state for entry in released: assert ledger.state(entry) == "released", f"step {step}: state reversed" - # retention stays bounded by the wave plus what is still in play - assert len(selector.rows) <= wave_size + ledger.live_count + reserve, ( + # retention stays bounded by the wave plus what is live -- a + # density-blocked candidate keeps its POSITION, never its row + assert len(selector.rows) <= wave_size + ledger.live_count, ( f"step {step}: retained {len(selector.rows)}" ) @@ -2460,3 +2459,56 @@ def test_take_tops_up_to_a_target_rather_than_adding_a_count(): assert selector.ledger.live_count == 4 assert len(selector.take(5)) == 1, "a wave refills exactly the freed slot" assert selector.ledger.live_count == 5 + + +def test_long_seen_run_does_not_discard_novel_rows_behind_the_cap( + recall_engine, monkeypatch +): + """DELTA-5: refunds are sequential, so no bounded buffer can hold the queue. + + A reserve sized to "no more slots than were admitted" assumed refunds + happen against SIMULTANEOUS admissions. Delta processing admits and + releases in sequence, so a long run of already-seen references can refund + far more slots than were ever live at once. With 100 ranked rows in one + session, a cap of 5 and the first 30 references seen, the reserve filled + with rows 6-30 and rows 31-100 were discarded outright -- the response came + back EMPTY even though rows 31-35 were novel, citable and still ranked. + + Candidates are already rank-ordered, so a blocked one is rediscoverable by + POSITION; nothing needs to be stored, and nothing can be dropped. + """ + _only_vector_arms(monkeypatch) + store_ids = _seed_citable_messages(recall_engine, 100, sessions=("session-a",)) + + seen: list[str] = [] + while len(seen) < 30: + page = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + include="verbatim", + scope_bias=0.0, + limit=10, + **({"seen_refs": list(seen)} if seen else {}), + ) + assert page["hits"], f"went empty after {len(seen)} seen refs" + seen.extend( + f"lcm:{hit['store_id']}:{hit['content_offset']}-" + f"{hit['content_offset'] + hit['content_returned_chars']}" + for hit in page["hits"] + ) + + after = _recall( + recall_engine, + monkeypatch, + detail="answer_ready", + include="verbatim", + scope_bias=0.0, + limit=10, + seen_refs=seen[:30], + ) + + assert after["hits"], "novel rows behind the cap must remain reachable" + assert not ({hit["exact_ref"] for hit in after["hits"]} & set(seen[:30])) + assert {hit["store_id"] for hit in after["hits"]} <= set(store_ids) + assert all(hit["content_offset"] is not None for hit in after["hits"]) diff --git a/tools.py b/tools.py index f6308b432..68200502b 100644 --- a/tools.py +++ b/tools.py @@ -3760,7 +3760,6 @@ def __init__( per_session_limit: int, expanded_limit: int, wave_size: int = _LCM_RECALL_STRICT_READ_WAVE, - defer_reserve: int = _LCM_RECALL_STRICT_DEFER_RESERVE, ) -> None: self._ordered = ordered self._engine = engine @@ -3768,12 +3767,14 @@ def __init__( self._per_session_limit = per_session_limit self._expanded_limit = expanded_limit self._wave_size = max(1, wave_size) - self._defer_reserve = max(0, defer_reserve) - self._deferred: list[dict[str, Any]] = [] - self._density_dropped = 0 + # Where to resume when a refund frees a slot: the position of the + # earliest density-blocked candidate still waiting, plus the identities + # of those candidates so they can be counted without being stored. + self._blocked_from: int | None = None + self._blocked: set[int] = set() self._examining = 0 self._prefetched_to = 0 - self._attempted: set[int] = set() + self._missing: set[int] = set() self.ledger = _LcmRecallSelectionLedger() self.rows: dict[int, dict[str, Any]] = {} self.batched_reads = 0 @@ -3781,11 +3782,11 @@ def __init__( @property def diversity_dropped(self) -> int: - """Candidates the density cap kept out, including those still deferred.""" - return self._density_dropped + len(self._deferred) + """Candidates the density cap is still keeping out of the response.""" + return len(self._blocked) def exhausted(self) -> bool: - return self._cursor >= len(self._ordered) and not self._deferred + return self._cursor >= len(self._ordered) def deliver(self, entry: dict[str, Any]) -> bool: return self.ledger.deliver(entry) @@ -3795,6 +3796,15 @@ def release(self, entry: dict[str, Any]) -> bool: released = self.ledger.release(entry) if released: self._forget(entry["hit"].get("store_id")) + # The freed slot belongs to the best-ranked candidate the cap kept + # out, which may be anywhere behind the cursor. Rewind to it rather + # than holding those candidates in a buffer: they are already in the + # ranking, so a POSITION is all that has to be remembered, and no + # bound on a buffer can then discard a valid row. + if self._blocked_from is not None: + self._cursor = min(self._cursor, self._blocked_from) + self._prefetched_to = min(self._prefetched_to, self._cursor) + self._blocked_from = None return released def _prefetch(self) -> bool: @@ -3814,7 +3824,7 @@ def _prefetch(self) -> bool: if hit.get("kind") == "summary" or store_id is None: continue store_id = int(store_id) - if store_id in self._attempted or store_id in wanted: + if store_id in self.rows or store_id in self._missing or store_id in wanted: continue wanted.append(store_id) if index == self._prefetched_to: @@ -3823,19 +3833,25 @@ def _prefetch(self) -> bool: if not wanted: return True self.batched_reads += 1 - # ATTEMPTED, not merely returned: a row the store does not have must be - # remembered as looked-for, or the walk cannot tell "not fetched yet" - # from "fetched and absent" and keeps calling for waves that can never - # contain it -- pulling in the rest of the corpus to reject one row. - self._attempted.update(wanted) - self.rows.update(self._engine._store.get_batch(wanted)) + fetched = self._engine._store.get_batch(wanted) + # A row the store does not have is remembered as MISSING, or the walk + # cannot tell "not fetched yet" from "fetched and absent" and keeps + # calling for waves that can never contain it. Rows merely evicted stay + # re-fetchable, which is what lets the walk revisit a candidate it + # skipped earlier without holding its row all along. + self._missing.update(sid for sid in wanted if sid not in fetched) + self.rows.update(fetched) return True def _row_for(self, store_id: Any) -> dict[str, Any] | None: if store_id is None: return None store_id = int(store_id) - while store_id not in self._attempted and self._prefetch(): + while ( + store_id not in self.rows + and store_id not in self._missing + and self._prefetch() + ): pass return self.rows.get(store_id) @@ -3846,10 +3862,6 @@ def _forget(self, store_id: Any) -> None: store_id = int(store_id) if self.ledger.holds_store(store_id): return - if any( - entry["hit"].get("store_id") == store_id for entry in self._deferred - ): - return self.rows.pop(store_id, None) def _verify(self, entry: dict[str, Any], *, hydratable: bool) -> bool: @@ -3900,48 +3912,55 @@ def _admit(self, entry: dict[str, Any]) -> None: store_id=hit.get("store_id"), ) - def _take_deferred(self) -> dict[str, Any] | None: - """Reconsider density-blocked candidates a refund may have unblocked.""" - for index, entry in enumerate(self._deferred): - session_key = self._session_key(entry["hit"]) - if self.ledger.session_count(session_key) < self._per_session_limit: - del self._deferred[index] - self._admit(entry) - return entry - return None - def _next_admissible(self) -> dict[str, Any] | None: - admitted = self._take_deferred() - if admitted is not None: - return admitted + """Walk forward to the next candidate that can be admitted right now. + + Entries the ledger already knows (live or released) and entries already + proved unciteable are skipped without being re-examined, so a rewind + costs a scan rather than a second round of reads. + """ while self._cursor < len(self._ordered): entry = self._ordered[self._cursor] self._examining = self._cursor self._cursor += 1 hit = entry["hit"] - hydratable = self.ledger.live_count < self._expanded_limit - if _lcm_recall_reference_shape(hit, hydratable=hydratable) is None: - self.unreferenced_dropped += 1 - self._forget(hit.get("store_id")) + if self.ledger.state(entry) is not None: continue - if not self._verify(entry, hydratable=hydratable): - self.unreferenced_dropped += 1 - self._forget(hit.get("store_id")) + if entry.get("_strict_rejected"): continue + hydratable = self.ledger.live_count < self._expanded_limit + # Verification is remembered per candidate, but only for the budget + # it was proved under: an in-budget candidate is proved by its row + # existing, which is a weaker claim than the excerpt check a + # post-budget slot needs. + if entry.get("_strict_verified") != hydratable: + if _lcm_recall_reference_shape(hit, hydratable=hydratable) is None or ( + not self._verify(entry, hydratable=hydratable) + ): + entry["_strict_rejected"] = True + self._blocked.discard(id(entry)) + self.unreferenced_dropped += 1 + self._forget(hit.get("store_id")) + continue + entry["_strict_verified"] = hydratable session_key = self._session_key(hit) if self.ledger.session_count(session_key) >= self._per_session_limit: - # Refundable, so hold it in rank order rather than discarding it - # -- but only up to the reserve, since no more slots can be - # refunded than were admitted. - if len(self._deferred) < self._defer_reserve: - self._deferred.append(entry) - else: - self._density_dropped += 1 - self._forget(hit.get("store_id")) + # Refundable, unlike a shape or verification failure, so the + # candidate is not consumed -- only its POSITION is remembered, + # and the row it is not using is released. It stays in the + # ranking, so no bound on a buffer can discard it. + if self._blocked_from is None: + self._blocked_from = self._cursor - 1 + self._blocked.add(id(entry)) + self._forget(hit.get("store_id")) continue + # Admission needs the row back: hydration reads from this snapshot, + # and this candidate may have had its row released while blocked. + self._row_for(hit.get("store_id")) + self._blocked.discard(id(entry)) self._admit(entry) return entry - return self._take_deferred() + return None def take(self, target: int) -> list[dict[str, Any]]: """Top the live set up to ``target`` verified candidates. From a8ada78b4b5605292fbf44ca0b0fe2149d4510e0 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 08:45:17 +0700 Subject: [PATCH 44/54] recall: expire cached state with its preconditions (delta-6 findings 1, 2) One family, one commit: state cached under conditions that no longer hold when the rewind brings the walk back to it. FINDING 1 -- rejection memory was not budget-specific. The two budgets prove different things: in-budget the delivered text is a window cut from the row, so the row existing IS the proof; post-budget the hit ships its own excerpt, which must be found at its offset. A candidate whose excerpt does not match its row therefore fails post-budget and passes in-budget -- but the rejection was remembered without its rule, so a refund that made the candidate in-budget again still skipped it and the response underfilled. Rejections are now keyed by the budget they were made under: an in-budget rejection means the row itself is unusable and stays permanent, a post-budget one bars only post-budget slots. The rejected set also replaces the counter, so re-examination cannot double count. A post-budget rejection is REVERSIBLE, so it now marks the resume position exactly as a density block does -- without that, the rewind never came back to it. FINDING 2 -- admission could proceed on a proof whose row was gone. The refetch result was discarded and the cached verification honoured, so a row evicted while blocked and then DELETED could still be admitted. Verification is no longer carried across examinations at all: it is only valid while the row it was proved against is still held, and a blocked candidate releases its row. Every examination re-proves against the row as refetched, which is also what makes a rewritten row safe, and admission needs no blind refetch because verification just left the row held. Extending the property generator to cover this family (post-budget rejections via candidates whose excerpt does not match their row, and delete-while-evicted) immediately found a third defect, introduced in delta-5 and previously outside the tested space: prefetch counted NEW ids rather than positions, so a window that mostly hit rows already held scanned far past itself to fill its quota and every rewind pulled in another wave -- retention grew without bound (50 rows against a bound of 37 by step 31). Read-ahead is now a span of POSITIONS and trims to the current window plus what is live, which is what actually enforces the wave + live bound. Tests: both reviewer repros; generator extended with the two transitions it could not previously produce, plus assertions that it actually reaches them (a deletion occurred, a post-budget rejection occurred, refetch pressure occurred) so the family cannot silently leave the tested space again. Mutations checked: budget-agnostic rejection, stale cached proof, no resume for post-budget rejections, and no row trimming are each caught -- the stale-proof mutation by the new repro AND by the generator. Counting ids instead of positions is NOT independently caught, because trimming bounds retention on its own; the positional window bounds work per prefetch rather than retention. Gate population unchanged: F35 replay 0/16 fail-closes, 25/25 rendered items, 400/400 hits publishing their offset, store-set churn +0/-0. Whole-suite failure set identical to baseline. --- tests/test_lcm_recall.py | 125 +++++++++++++++++++++++++++++++- tools.py | 151 +++++++++++++++++++++++++-------------- 2 files changed, 219 insertions(+), 57 deletions(-) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index 553778e5e..7c8b0662e 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -1415,6 +1415,10 @@ def get_batch(self, store_ids): self.batch_calls.append(list(store_ids)) return {sid: self._rows[sid] for sid in store_ids if sid in self._rows} + def drop(self, store_id): + """Delete a row, as supported cleanup does mid-request.""" + self._rows.pop(store_id, None) + def _stub_engine(store_ids): return SimpleNamespace(_store=_StubStore(store_ids)) @@ -2371,9 +2375,20 @@ def test_ledger_invariants_hold_under_a_randomized_transition_sequence(): rng = random.Random(20260729) per_session_limit = 3 wave_size = 16 + # expanded_limit > 0 so the SAME candidate can be judged in-budget or + # post-budget as the live count moves across the threshold -- the transition + # that makes a rejection's rule matter. A third of the candidates carry an + # excerpt that does not match their row: admissible in-budget (where the + # text is cut from the row) and not post-budget (where it must be found at + # its offset), so post-budget rejections are inside the generated space. + expanded_limit = 4 sessions = [f"session-{index % 7}" for index in range(1, 121)] ordered = [ - _message_entry(index, session_id=sessions[index - 1]) + _message_entry( + index, + session_id=sessions[index - 1], + snippet="not in the row" if index % 3 == 0 else None, + ) for index in range(1, 121) ] engine = _stub_engine(range(1, 121)) @@ -2381,9 +2396,10 @@ def test_ledger_invariants_hold_under_a_randomized_transition_sequence(): ordered, engine=engine, per_session_limit=per_session_limit, - expanded_limit=0, + expanded_limit=expanded_limit, wave_size=wave_size, ) + deleted: set[int] = set() ledger = selector.ledger admitted: list[dict] = [] delivered: list[dict] = [] @@ -2410,9 +2426,31 @@ def check(step): for step in range(200): action = rng.random() - if action < 0.4: + if action < 0.35: for entry in selector.take(ledger.live_count + rng.randint(1, 4)): + store_id = entry["hit"]["store_id"] + # Admission must rest on the row as it stands NOW: the walk + # holds it, and it was never deleted behind the proof. + assert store_id in selector.rows, ( + f"step {step}: admitted {store_id} without holding its row" + ) + assert store_id not in deleted, ( + f"step {step}: admitted {store_id} on a stale proof" + ) admitted.append(entry) + elif action < 0.45: + # Delete a row nothing live depends on -- the eviction-then-refetch + # path, where a blocked candidate's row disappears while it waits. + candidates = [ + sid + for sid in range(1, 121) + if sid not in deleted and not ledger.holds_store(sid) + ] + if candidates: + victim = rng.choice(candidates) + engine._store.drop(victim) + selector.rows.pop(victim, None) + deleted.add(victim) elif action < 0.6 and admitted: entry = rng.choice(admitted) if ledger.deliver(entry): @@ -2430,6 +2468,13 @@ def check(step): assert delivered, "the sequence must actually deliver something" assert set(map(id, ledger.delivered_entries())) == set(map(id, delivered)) assert ledger.double_releases > 0, "the sequence must exercise double release" + # The generator must actually reach the family this round was about, or the + # family sits outside the tested space and the property proves nothing here. + assert deleted, "no row was ever deleted mid-walk" + assert any( + under is False for under in selector._rejected.values() + ), "no post-budget rejection was generated" + assert selector.batched_reads > wave_size // 4, "no refetch pressure generated" def test_take_tops_up_to_a_target_rather_than_adding_a_count(): @@ -2512,3 +2557,77 @@ def test_long_seen_run_does_not_discard_novel_rows_behind_the_cap( assert not ({hit["exact_ref"] for hit in after["hits"]} & set(seen[:30])) assert {hit["store_id"] for hit in after["hits"]} <= set(store_ids) assert all(hit["content_offset"] is not None for hit in after["hits"]) + + +# -- Delta-6 review: cache coherence across a rewind --------------------------- + + +def test_post_budget_rejection_does_not_bar_a_later_in_budget_admission(): + """DELTA-6 FINDING 1: a rejection is only as durable as its rule. + + The two budgets prove different things. In-budget, the delivered text is a + window cut from the row, so the row existing IS the proof. Post-budget, the + hit ships its own excerpt, which must be found at its offset. A candidate + carrying an excerpt that does not match its row therefore fails post-budget + and passes in-budget -- but the rejection was remembered without its rule, + so once a refund made the candidate in-budget again it stayed skipped and + the response underfilled. + """ + # Excerpt does not match the row: post-budget it cannot be cited, in-budget + # it can, because hydration cuts the text from the row itself. + ordered = [ + _message_entry(1, session_id="session-a"), + _message_entry(2, session_id="session-b", snippet="not in the row"), + ] + engine = _stub_engine([1, 2]) + selector = lcm_tools._LcmRecallStrictSelector( + ordered, engine=engine, per_session_limit=5, expanded_limit=1 + ) + + first = selector.take(1) + assert [e["hit"]["store_id"] for e in first] == [1] + # Live == expanded_limit, so candidate 2 is judged post-budget and rejected. + assert selector.take(2) == [] + assert selector.unreferenced_dropped == 1 + + # The refund puts the budget back; candidate 2 is now admissible in-budget. + assert selector.release(first[0]) is True + regained = selector.take(1) + assert [e["hit"]["store_id"] for e in regained] == [2], ( + "a post-budget rejection must not bar an in-budget admission" + ) + assert selector.unreferenced_dropped == 0 + + +def test_row_deleted_while_blocked_is_reverified_before_admission(): + """DELTA-6 FINDING 2: cached proof is void once its row is let go. + + A blocked candidate releases its row. If that row is then deleted, a refund + rewinding to the candidate must NOT admit it on the strength of the earlier + verification -- the bytes it was proved against are gone. + """ + ordered = [ + _message_entry(1, session_id="session-a"), + _message_entry(2, session_id="session-a"), + _message_entry(3, session_id="session-b"), + ] + engine = _stub_engine([1, 2, 3]) + selector = lcm_tools._LcmRecallStrictSelector( + ordered, engine=engine, per_session_limit=1, expanded_limit=8 + ) + + taken = selector.take(2) + assert [e["hit"]["store_id"] for e in taken] == [1, 3] + # Candidate 2 was verified, then blocked on session-a's cap, so its row was + # released. Delete it from the store while it waits. + assert 2 not in selector.rows + engine._store.drop(2) + + assert selector.release(taken[0]) is True + regained = selector.take(2) + + assert [e["hit"]["store_id"] for e in regained] == [], ( + "a deleted row must be re-proved, not admitted from a stale proof" + ) + assert not selector.ledger.holds_store(2) + assert selector.unreferenced_dropped == 1 diff --git a/tools.py b/tools.py index 68200502b..74a7bcf4d 100644 --- a/tools.py +++ b/tools.py @@ -3767,18 +3767,26 @@ def __init__( self._per_session_limit = per_session_limit self._expanded_limit = expanded_limit self._wave_size = max(1, wave_size) - # Where to resume when a refund frees a slot: the position of the - # earliest density-blocked candidate still waiting, plus the identities - # of those candidates so they can be counted without being stored. - self._blocked_from: int | None = None + # Where to resume when a refund frees a slot: the earliest position + # holding a candidate that was skipped for a REVERSIBLE reason -- the + # density cap, or a post-budget rejection that an in-budget slot would + # judge by a weaker rule. Only the position is kept; the candidates stay + # in the ranking. + self._resume_from: int | None = None self._blocked: set[int] = set() + # id(entry) -> the budget the rejection was made under. + self._rejected: dict[int, bool] = {} self._examining = 0 self._prefetched_to = 0 self._missing: set[int] = set() self.ledger = _LcmRecallSelectionLedger() self.rows: dict[int, dict[str, Any]] = {} self.batched_reads = 0 - self.unreferenced_dropped = 0 + + @property + def unreferenced_dropped(self) -> int: + """Candidates currently held out for want of a validated reference.""" + return len(self._rejected) @property def diversity_dropped(self) -> int: @@ -3801,48 +3809,66 @@ def release(self, entry: dict[str, Any]) -> bool: # than holding those candidates in a buffer: they are already in the # ranking, so a POSITION is all that has to be remembered, and no # bound on a buffer can then discard a valid row. - if self._blocked_from is not None: - self._cursor = min(self._cursor, self._blocked_from) + if self._resume_from is not None: + self._cursor = min(self._cursor, self._resume_from) self._prefetched_to = min(self._prefetched_to, self._cursor) - self._blocked_from = None + self._resume_from = None return released def _prefetch(self) -> bool: - """Read the next wave of candidate rows in ONE batch. + """Read the next WINDOW of candidate rows in ONE batch. - Starts at the candidate being examined -- NOT after it -- so the row the - walk needs right now is always inside the wave it triggers. + The window is a span of POSITIONS, not a quota of new ids. Counting new + ids instead lets a window that mostly hits rows already held scan far + past itself to fill its quota, so every rewind pulls in another wave and + retention grows without bound. Advancing by position keeps read-ahead -- + and therefore retention -- inside a single window. + + Starts at the candidate being examined, NOT after it, so the row the walk + needs right now is always inside the window it triggers. """ if self._engine is None: return False - wanted: list[int] = [] - index = max(self._examining, self._prefetched_to) - while index < len(self._ordered) and len(wanted) < self._wave_size: + start = max(self._examining, self._prefetched_to) + if start >= len(self._ordered): + return False + window: list[int] = [] + index = start + while index < len(self._ordered) and index - start < self._wave_size: hit = self._ordered[index]["hit"] index += 1 store_id = hit.get("store_id") if hit.get("kind") == "summary" or store_id is None: continue store_id = int(store_id) - if store_id in self.rows or store_id in self._missing or store_id in wanted: - continue - wanted.append(store_id) - if index == self._prefetched_to: - return False + if store_id not in window: + window.append(store_id) self._prefetched_to = index - if not wanted: - return True - self.batched_reads += 1 - fetched = self._engine._store.get_batch(wanted) - # A row the store does not have is remembered as MISSING, or the walk - # cannot tell "not fetched yet" from "fetched and absent" and keeps - # calling for waves that can never contain it. Rows merely evicted stay - # re-fetchable, which is what lets the walk revisit a candidate it - # skipped earlier without holding its row all along. - self._missing.update(sid for sid in wanted if sid not in fetched) - self.rows.update(fetched) + wanted = [ + store_id + for store_id in window + if store_id not in self.rows and store_id not in self._missing + ] + if wanted: + self.batched_reads += 1 + fetched = self._engine._store.get_batch(wanted) + # A row the store does not have is remembered as MISSING, or the walk + # cannot tell "not fetched yet" from "fetched and absent" and keeps + # calling for windows that can never contain it. Rows merely evicted + # stay re-fetchable, which is what lets the walk revisit a candidate + # it skipped earlier without holding its row all along. + self._missing.update(sid for sid in wanted if sid not in fetched) + self.rows.update(fetched) + self._trim_rows(window) return True + def _trim_rows(self, window: list[int]) -> None: + """Hold only the current window and whatever is live.""" + keep = set(window) + for store_id in list(self.rows): + if store_id not in keep and not self.ledger.holds_store(store_id): + del self.rows[store_id] + def _row_for(self, store_id: Any) -> dict[str, Any] | None: if store_id is None: return None @@ -3912,12 +3938,17 @@ def _admit(self, entry: dict[str, Any]) -> None: store_id=hit.get("store_id"), ) + def _mark_resume(self) -> None: + """Remember the earliest position a refund would have to come back to.""" + position = self._cursor - 1 + if self._resume_from is None or position < self._resume_from: + self._resume_from = position + def _next_admissible(self) -> dict[str, Any] | None: """Walk forward to the next candidate that can be admitted right now. - Entries the ledger already knows (live or released) and entries already - proved unciteable are skipped without being re-examined, so a rewind - costs a scan rather than a second round of reads. + Entries the ledger already knows (live or released) are skipped, as are + entries whose rejection still applies under the current budget. """ while self._cursor < len(self._ordered): entry = self._ordered[self._cursor] @@ -3926,38 +3957,50 @@ def _next_admissible(self) -> dict[str, Any] | None: hit = entry["hit"] if self.ledger.state(entry) is not None: continue - if entry.get("_strict_rejected"): - continue hydratable = self.ledger.live_count < self._expanded_limit - # Verification is remembered per candidate, but only for the budget - # it was proved under: an in-budget candidate is proved by its row - # existing, which is a weaker claim than the excerpt check a - # post-budget slot needs. - if entry.get("_strict_verified") != hydratable: - if _lcm_recall_reference_shape(hit, hydratable=hydratable) is None or ( - not self._verify(entry, hydratable=hydratable) - ): - entry["_strict_rejected"] = True - self._blocked.discard(id(entry)) - self.unreferenced_dropped += 1 - self._forget(hit.get("store_id")) - continue - entry["_strict_verified"] = hydratable + # A rejection is only as durable as the rule that produced it. An + # IN-BUDGET rejection means the row itself is unusable -- missing, + # empty, or not a message at all -- which no later budget can undo. + # A POST-BUDGET rejection only means the candidate's own excerpt did + # not check out, and that same candidate is still admissible + # in-budget, where the delivered text is a window cut from the row + # rather than an excerpt it carried. Skipping it there would underfill + # against a rule that no longer applies. + rejected_under = self._rejected.get(id(entry)) + if rejected_under is True or (rejected_under is False and not hydratable): + continue + # Verification is deliberately NOT carried across examinations. It is + # only valid while the row it was proved against is still held, and a + # blocked candidate releases its row -- which may then be rewritten or + # deleted before a refund brings the walk back. Re-proving against the + # row as REFETCHED is what makes the rewind safe. + if _lcm_recall_reference_shape(hit, hydratable=hydratable) is None or ( + not self._verify(entry, hydratable=hydratable) + ): + self._rejected[id(entry)] = hydratable + self._blocked.discard(id(entry)) + # Reversible: judged by the post-budget rule, this candidate is + # still admissible in-budget, so a refund must bring the walk + # back to it exactly as it does for a density block. + if not hydratable: + self._mark_resume() + self._forget(hit.get("store_id")) + continue session_key = self._session_key(hit) if self.ledger.session_count(session_key) >= self._per_session_limit: # Refundable, unlike a shape or verification failure, so the # candidate is not consumed -- only its POSITION is remembered, # and the row it is not using is released. It stays in the # ranking, so no bound on a buffer can discard it. - if self._blocked_from is None: - self._blocked_from = self._cursor - 1 + self._mark_resume() self._blocked.add(id(entry)) self._forget(hit.get("store_id")) continue - # Admission needs the row back: hydration reads from this snapshot, - # and this candidate may have had its row released while blocked. - self._row_for(hit.get("store_id")) + # No blind refetch here: verification above just proved this + # candidate against the row as it stands and left that row held, so + # hydration reads the same bytes the admission was granted on. self._blocked.discard(id(entry)) + self._rejected.pop(id(entry), None) self._admit(entry) return entry return None From 60491577e6365e4b54c668864f72764409804e65 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 09:01:46 +0700 Subject: [PATCH 45/54] recall: derive the resume point instead of storing it (delta-7) FINDING 1 -- a stored resume cursor is the same family this cycle has been closing everywhere else: state that outlives its precondition. The first rewind consumed _resume_from, and a later walk that SKIPPED an already-rejected post-budget candidate never re-armed it, so once the cursor had stepped past that candidate no subsequent refund could reach it -- permanently unreachable though admissible again. _resume_from is gone. On a refund the walk resumes from the earliest NON-SETTLED position, derived from the ledger and the rejection map: settled means the ledger already knows the entry (admitted, delivered or released, none of which return to the pool) or it was rejected IN-BUDGET, which means the row itself is unusable. Everything else -- density-blocked, post-budget rejected -- is revisitable by definition, because the only thing in its way is a budget a refund can give back. A _settled_prefix hint keeps the scan cheap; it is sound because settledness is monotone, and it is only ever a hint -- the resume point itself cannot go stale because it is never stored. Every piece of walk state is now either in the ledger or derived from it. FINDING 2 -- the generator's reach assertions proved aggregates, not interleavings. "Some row was deleted" and "some post-budget rejection occurred" can both hold in one run without those events ever meeting in a single candidate's life, which is exactly where these bugs live. Candidates now carry per-candidate HISTORIES -- status transitions plus evicted/row-deleted/refund markers -- and the assertions are subsequence checks over them: at least one candidate must have lived blocked -> evicted -> row-deleted -> refund -> re-examined, and at least one post-budget-rejected -> refund -> admitted. The histories immediately proved their worth: the second transition was UNREACHABLE under the old action mix, because the live count rarely fell below the hydration budget, so a post-budget rejection was only ever re-judged post-budget. A drain action (hand back every live slot at once) makes it reachable. The aggregate assertions had been passing over that gap. Regression: the partial-refund repro, built on the interleaving that actually distinguishes -- an earlier density block takes the single stored resume slot, the rewind spends it, the next walk steps over the rejected candidate, and only then do the remaining refunds land. Mutation-checked: reinstating a stored resume cursor turns it red. Budget-agnostic rejection and treating a post-budget rejection as settled each take down three tests including the generator. Gate population unchanged: F35 replay 0/16 fail-closes, 25/25 rendered items, 400/400 hits publishing their offset, store-set churn +0/-0. Whole-suite failure set identical to baseline. --- tests/test_lcm_recall.py | 202 ++++++++++++++++++++++++++++++++------- tools.py | 60 +++++++----- 2 files changed, 200 insertions(+), 62 deletions(-) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index 7c8b0662e..2275a579e 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -2363,12 +2363,34 @@ def test_released_entry_stops_pinning_its_row(): ) +def _has_subsequence(history, pattern): + """True when `pattern` appears in order (not necessarily adjacently).""" + index = 0 + for event in history: + if index < len(pattern) and pattern[index](event): + index += 1 + return index == len(pattern) + + +def _is(name): + return lambda event: event == name + + +def _startswith(prefix): + return lambda event: event.startswith(prefix) + + def test_ledger_invariants_hold_under_a_randomized_transition_sequence(): - """Property check: no admit/deliver/release interleaving can break the books. + """Property check: no interleaving of the lifecycle can break the books. + + Each review round found a different piece of walk state leaking at a + different seam, so the accounting is exercised as a state machine rather + than only at the seams already known to have failed. - Three rounds of review each found a different counter leaking at a different - seam, so the accounting is exercised as a state machine rather than only at - the seams already known to have failed. + The reach assertions are on per-candidate HISTORIES, not on aggregate + counters: a seed can easily satisfy "some row was deleted" and "some + post-budget rejection happened" without those ever meeting in one + candidate's life, which is precisely where the bugs have been living. """ import random @@ -2376,11 +2398,10 @@ def test_ledger_invariants_hold_under_a_randomized_transition_sequence(): per_session_limit = 3 wave_size = 16 # expanded_limit > 0 so the SAME candidate can be judged in-budget or - # post-budget as the live count moves across the threshold -- the transition - # that makes a rejection's rule matter. A third of the candidates carry an - # excerpt that does not match their row: admissible in-budget (where the - # text is cut from the row) and not post-budget (where it must be found at - # its offset), so post-budget rejections are inside the generated space. + # post-budget as the live count crosses the threshold -- the transition that + # makes a rejection's rule matter. A third of the candidates carry an excerpt + # that does not match their row: admissible in-budget (where the text is cut + # from the row) and not post-budget (where it must be found at its offset). expanded_limit = 4 sessions = [f"session-{index % 7}" for index in range(1, 121)] ordered = [ @@ -2399,38 +2420,71 @@ def test_ledger_invariants_hold_under_a_randomized_transition_sequence(): expanded_limit=expanded_limit, wave_size=wave_size, ) - deleted: set[int] = set() ledger = selector.ledger admitted: list[dict] = [] delivered: list[dict] = [] released: list[dict] = [] + deleted: set[int] = set() + by_store = {entry["hit"]["store_id"]: entry for entry in ordered} + history: dict[int, list[str]] = {id(entry): ["untouched"] for entry in ordered} + + def status_of(entry): + state = ledger.state(entry) + if state is not None: + return state + if id(entry) in selector._blocked: + return "blocked" + rejected = selector._rejected.get(id(entry)) + if rejected is True: + return "rejected-in" + if rejected is False: + return "rejected-post" + return "untouched" + + def record(entry, event): + log = history[id(entry)] + if log[-1] != event: + log.append(event) + + def observe(): + """Append each candidate's status change, and note released rows.""" + for entry in ordered: + status = status_of(entry) + record(entry, status) + if status in ("blocked", "rejected-post"): + if entry["hit"]["store_id"] not in selector.rows: + record(entry, "evicted") def check(step): - # per-session live count never exceeds the cap for key in {selector._session_key(e["hit"]) for e in ordered}: assert ledger.session_count(key) <= per_session_limit, f"step {step}: {key}" - # refunds are never double-applied: the books agree with the records live = [e for e in admitted if ledger.state(e) in ("admitted", "delivered")] assert ledger.live_count == len(live), f"step {step}: live drift" assert sum(ledger._session_counts.values()) == ledger.live_count, ( f"step {step}: session totals drift" ) - # a released entry never returns to a live state for entry in released: assert ledger.state(entry) == "released", f"step {step}: state reversed" - # retention stays bounded by the wave plus what is live -- a - # density-blocked candidate keeps its POSITION, never its row assert len(selector.rows) <= wave_size + ledger.live_count, ( f"step {step}: retained {len(selector.rows)}" ) + def refund(entry): + if selector.release(entry): + released.append(entry) + # A refund happened while every unsettled candidate waited. + for other in ordered: + if not selector._is_settled(other): + record(other, "refund") + return True + return False + for step in range(200): action = rng.random() - if action < 0.35: - for entry in selector.take(ledger.live_count + rng.randint(1, 4)): + if action < 0.32: + for entry in selector.take(ledger.live_count + rng.randint(1, 6)): store_id = entry["hit"]["store_id"] - # Admission must rest on the row as it stands NOW: the walk - # holds it, and it was never deleted behind the proof. + # Admission must rest on the row as it stands NOW. assert store_id in selector.rows, ( f"step {step}: admitted {store_id} without holding its row" ) @@ -2438,43 +2492,68 @@ def check(step): f"step {step}: admitted {store_id} on a stale proof" ) admitted.append(entry) - elif action < 0.45: - # Delete a row nothing live depends on -- the eviction-then-refetch - # path, where a blocked candidate's row disappears while it waits. - candidates = [ + elif action < 0.42: + # Prefer deleting a row whose candidate is waiting with its row + # already released -- the blocked -> evicted -> deleted path. + waiting = [ + sid + for sid, entry in by_store.items() + if sid not in deleted + and not ledger.holds_store(sid) + and status_of(entry) in ("blocked", "rejected-post") + and sid not in selector.rows + ] + pool = waiting or [ sid - for sid in range(1, 121) + for sid in by_store if sid not in deleted and not ledger.holds_store(sid) ] - if candidates: - victim = rng.choice(candidates) + if pool: + victim = rng.choice(pool) engine._store.drop(victim) selector.rows.pop(victim, None) deleted.add(victim) - elif action < 0.6 and admitted: + record(by_store[victim], "row-deleted") + elif action < 0.54 and admitted: entry = rng.choice(admitted) if ledger.deliver(entry): delivered.append(entry) - elif action < 0.9 and admitted: - entry = rng.choice(admitted) - if selector.release(entry): - released.append(entry) + elif action < 0.80 and admitted: + refund(rng.choice(admitted)) + elif action < 0.90 and admitted: + # DRAIN: hand back every live slot at once. Without this the live + # count rarely falls under the hydration budget, so a post-budget + # rejection is only ever re-judged post-budget and the + # rejected-then-revisited-in-budget transition stays unreachable. + for entry in list(admitted): + if ledger.state(entry) == "admitted": + refund(entry) elif released: before = ledger.live_count assert selector.release(rng.choice(released)) is False assert ledger.live_count == before, f"step {step}: double release refunded" + observe() check(step) assert delivered, "the sequence must actually deliver something" assert set(map(id, ledger.delivered_entries())) == set(map(id, delivered)) assert ledger.double_releases > 0, "the sequence must exercise double release" - # The generator must actually reach the family this round was about, or the - # family sits outside the tested space and the property proves nothing here. - assert deleted, "no row was ever deleted mid-walk" + + # -- Reach, proved on HISTORIES: the family must be inside the tested space, + # and aggregate counters cannot show that these events ever MET. -- + logs = list(history.values()) assert any( - under is False for under in selector._rejected.values() - ), "no post-budget rejection was generated" - assert selector.batched_reads > wave_size // 4, "no refetch pressure generated" + _has_subsequence( + log, + [_is("blocked"), _is("evicted"), _is("row-deleted"), _is("refund"), + _startswith("rejected")], + ) + for log in logs + ), "no candidate lived blocked -> evicted -> deleted -> refund -> re-examined" + assert any( + _has_subsequence(log, [_is("rejected-post"), _is("refund"), _is("admitted")]) + for log in logs + ), "no candidate was post-budget rejected then revisited in-budget" def test_take_tops_up_to_a_target_rather_than_adding_a_count(): @@ -2631,3 +2710,52 @@ def test_row_deleted_while_blocked_is_reverified_before_admission(): ) assert not selector.ledger.holds_store(2) assert selector.unreferenced_dropped == 1 + + +def test_reversible_rejection_stays_reachable_after_a_partial_refund(): + """DELTA-7: the resume point must be DERIVED, not remembered. + + A stored resume cursor is state that outlives its precondition. The first + rewind consumes it, and a later walk that SKIPS an already-rejected + post-budget candidate never re-arms it -- so once the cursor has moved past + that candidate, no subsequent refund can reach it and it is permanently + unreachable although it is admissible again. + + The interleaving that exposes it: a density block earlier in the ranking + takes the one stored resume slot, the rewind spends it, the next walk steps + over the rejected candidate, and only then do the remaining refunds land. + """ + # 2 carries an excerpt that does not match its row: rejected post-budget, + # admissible in-budget where the text is cut from the row instead. + ordered = [ + _message_entry(1, session_id="session-a"), + _message_entry(2, session_id="session-a"), + _message_entry(3, session_id="session-b", snippet="not in the row"), + _message_entry(4, session_id="session-c"), + _message_entry(5, session_id="session-d"), + ] + engine = _stub_engine([1, 2, 3, 4, 5]) + selector = lcm_tools._LcmRecallStrictSelector( + ordered, engine=engine, per_session_limit=1, expanded_limit=1 + ) + + first = selector.take(5) + assert [e["hit"]["store_id"] for e in first] == [1, 4, 5] + assert id(ordered[1]) in selector._blocked, "2 is held by the density cap" + assert selector._rejected.get(id(ordered[2])) is False, "3 is post-budget rejected" + + # The density block takes the single stored resume slot; this refund spends it. + assert selector.release(first[0]) is True + assert [e["hit"]["store_id"] for e in selector.take(3)] == [2] + + # This walk stepped OVER candidate 3 without re-arming any stored resume. + assert selector.release(first[1]) is True + assert selector.take(3) == [] + assert selector.release(first[2]) is True + assert selector.release(ordered[1]) is True + assert selector.ledger.live_count == 0 + + regained = selector.take(1) + assert [e["hit"]["store_id"] for e in regained] == [3], ( + "a revisitable candidate must never become unreachable" + ) diff --git a/tools.py b/tools.py index 74a7bcf4d..560c6dbb1 100644 --- a/tools.py +++ b/tools.py @@ -3767,12 +3767,11 @@ def __init__( self._per_session_limit = per_session_limit self._expanded_limit = expanded_limit self._wave_size = max(1, wave_size) - # Where to resume when a refund frees a slot: the earliest position - # holding a candidate that was skipped for a REVERSIBLE reason -- the - # density cap, or a post-budget rejection that an in-budget slot would - # judge by a weaker rule. Only the position is kept; the candidates stay - # in the ranking. - self._resume_from: int | None = None + # How far the ranking is known to be SETTLED. Monotone: an entry the + # ledger knows, or one rejected in-budget, can never become admissible + # again, so this only ever advances. It is a scan hint, never a source + # of truth -- the resume point itself is derived on demand. + self._settled_prefix = 0 self._blocked: set[int] = set() # id(entry) -> the budget the rejection was made under. self._rejected: dict[int, bool] = {} @@ -3804,15 +3803,15 @@ def release(self, entry: dict[str, Any]) -> bool: released = self.ledger.release(entry) if released: self._forget(entry["hit"].get("store_id")) - # The freed slot belongs to the best-ranked candidate the cap kept - # out, which may be anywhere behind the cursor. Rewind to it rather - # than holding those candidates in a buffer: they are already in the - # ranking, so a POSITION is all that has to be remembered, and no - # bound on a buffer can then discard a valid row. - if self._resume_from is not None: - self._cursor = min(self._cursor, self._resume_from) + # Resume from the earliest candidate that could still be admitted. + # DERIVED, never stored: a remembered resume position is state that + # outlives its precondition -- cleared by one rewind, not re-armed by + # the next skip, and a candidate silently becomes unreachable. Asking + # the ranking costs a scan and cannot go stale. + resume = self._earliest_revisitable() + if resume < self._cursor: + self._cursor = resume self._prefetched_to = min(self._prefetched_to, self._cursor) - self._resume_from = None return released def _prefetch(self) -> bool: @@ -3938,11 +3937,28 @@ def _admit(self, entry: dict[str, Any]) -> None: store_id=hit.get("store_id"), ) - def _mark_resume(self) -> None: - """Remember the earliest position a refund would have to come back to.""" - position = self._cursor - 1 - if self._resume_from is None or position < self._resume_from: - self._resume_from = position + def _is_settled(self, entry: dict[str, Any]) -> bool: + """True when this candidate can never be admitted, whatever happens next. + + Terminal for exactly two reasons: the ledger already knows it (admitted, + delivered or released -- none of which return to the pool), or it was + rejected IN-BUDGET, which means the row itself is unusable. Everything + else -- density-blocked, post-budget rejected -- is revisitable by + definition, because the only thing standing in its way is a budget a + refund can give back. + """ + return ( + self.ledger.state(entry) is not None + or self._rejected.get(id(entry)) is True + ) + + def _earliest_revisitable(self) -> int: + """Position of the first candidate a refund could still make admissible.""" + index = self._settled_prefix + while index < len(self._ordered) and self._is_settled(self._ordered[index]): + index += 1 + self._settled_prefix = index + return index def _next_admissible(self) -> dict[str, Any] | None: """Walk forward to the next candidate that can be admitted right now. @@ -3979,11 +3995,6 @@ def _next_admissible(self) -> dict[str, Any] | None: ): self._rejected[id(entry)] = hydratable self._blocked.discard(id(entry)) - # Reversible: judged by the post-budget rule, this candidate is - # still admissible in-budget, so a refund must bring the walk - # back to it exactly as it does for a density block. - if not hydratable: - self._mark_resume() self._forget(hit.get("store_id")) continue session_key = self._session_key(hit) @@ -3992,7 +4003,6 @@ def _next_admissible(self) -> dict[str, Any] | None: # candidate is not consumed -- only its POSITION is remembered, # and the row it is not using is released. It stays in the # ranking, so no bound on a buffer can discard it. - self._mark_resume() self._blocked.add(id(entry)) self._forget(hit.get("store_id")) continue From 247f81177874eda1d942aa9a6fa2088ab9b19fc7 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 11:19:53 +0700 Subject: [PATCH 46/54] bench-tooling: round-1 review fixes (bootstrap cleanup, trace cache, guards, pricing import, probe limits) CodeRabbit round-1 items 11-21 on PR #175: 8 fixed, 2 declined (deliberate self-contained duplication in replay drivers). Reviewed by orchestrator. --- benchmarking/h3_composition_replay.py | 21 ++- benchmarking/h5_recall_replay.py | 4 +- benchmarking/state_embedding_backfill.py | 200 +++++++++++++---------- 3 files changed, 131 insertions(+), 94 deletions(-) diff --git a/benchmarking/h3_composition_replay.py b/benchmarking/h3_composition_replay.py index e71853749..dbc14ef36 100644 --- a/benchmarking/h3_composition_replay.py +++ b/benchmarking/h3_composition_replay.py @@ -37,6 +37,7 @@ import statistics import sys import time +from functools import lru_cache from pathlib import Path from typing import Any @@ -85,11 +86,22 @@ def _bootstrap_package(repo_root: Path) -> Any: setattr(mod, py_file.stem, sub_mod) try: sub_spec.loader.exec_module(sub_mod) - except Exception: - pass # unrelated modules (engine needs `agent`) may fail; ignore + except Exception as exc: + sys.modules.pop(sub_name, None) + if getattr(mod, py_file.stem, None) is sub_mod: + delattr(mod, py_file.stem) + print( + f"warning: failed to bootstrap {sub_name}: {exc!r}", + file=sys.stderr, + ) return mod +@lru_cache(maxsize=None) +def _read_trace(path: str) -> dict[str, Any]: + return json.loads(Path(path).read_text()) + + def _question_text(question_field: Any) -> str: """The exact query string the reader harness passes to ``query()``. @@ -169,7 +181,7 @@ def trace(self, qid: str) -> dict[str, Any]: self.h31 / "query_traces" / domain / "query_traces" / qid / "hermes_lcm_semantic_telemetry.json" ) - return json.loads(path.read_text()) + return _read_trace(str(path)) def _injected_ranks(self, qid: str) -> list[tuple[int, float]]: ranks = sorted(self.trace(qid)["source_candidate_ranks"], key=lambda r: r["rank"]) @@ -356,7 +368,7 @@ def measure_latency(ctx: ReplayContext, sample_qids: list[str], knob_kwargs: dic return { "p50_ms": _percentile(latencies, 50), "p95_ms": _percentile(latencies, 95), - "mean_ms": statistics.fmean(latencies), + "mean_ms": statistics.fmean(latencies) if latencies else 0.0, } @@ -437,6 +449,7 @@ def main() -> int: "baseline_latency": baseline_latency, "sweep": results, } + args.out.parent.mkdir(parents=True, exist_ok=True) args.out.write_text(json.dumps(payload, indent=2)) print(f"\nwrote {args.out}") return 0 if golden["passed"] == golden["total"] else 1 diff --git a/benchmarking/h5_recall_replay.py b/benchmarking/h5_recall_replay.py index 7a366f358..60ba80c04 100644 --- a/benchmarking/h5_recall_replay.py +++ b/benchmarking/h5_recall_replay.py @@ -110,6 +110,8 @@ def traj_source_id(self, domain: str, trajectory_id: str) -> int: "SELECT source_id FROM lcm_trajectory_sources WHERE trajectory_id = ?", (trajectory_id,), ).fetchone() + if row is None: + raise RuntimeError(f"{domain}/{trajectory_id} missing from frozen DB") return int(row[0]) # -- exact pool membership -------------------------------------------- @@ -168,7 +170,7 @@ def seed_split(self) -> dict[str, Any]: any_pooled = True if expression: source_id = self.traj_source_id(domain, target["trajectory_id"]) - rows = store._fts_rows(expression, 100000, source_ids=[source_id]) + rows = store._fts_rows(expression, 1, source_ids=[source_id]) if rows: any_match = True per_case[qid] = ( diff --git a/benchmarking/state_embedding_backfill.py b/benchmarking/state_embedding_backfill.py index fd3bdcd9f..1dd6e5b73 100644 --- a/benchmarking/state_embedding_backfill.py +++ b/benchmarking/state_embedding_backfill.py @@ -30,9 +30,10 @@ _REPO_ROOT = Path(__file__).resolve().parent.parent -# voyage-4 document-embedding price ($/1M tokens); mirrors command.py's -# _VOYAGE_USD_PER_MILLION_TOKENS (the table the #141 sizing used). -_VOYAGE_USD_PER_MILLION_TOKENS = { +# WARNING: fallback for standalone contexts where command.py's canonical +# _VOYAGE_USD_PER_MILLION_TOKENS table cannot be imported. Keep this duplicate +# synchronized; drift here changes spend estimates. +_FALLBACK_VOYAGE_USD_PER_MILLION_TOKENS = { "voyage-4-large": 0.12, "voyage-4": 0.06, "voyage-4-lite": 0.02, @@ -43,6 +44,7 @@ def _bootstrap_package(repo_root: Path) -> Any: + """Register the plugin dir as the ``hermes_lcm`` package (mirrors conftest).""" pkg = "hermes_lcm" if pkg in sys.modules: return sys.modules[pkg] @@ -72,11 +74,30 @@ def _bootstrap_package(repo_root: Path) -> Any: setattr(mod, py_file.stem, sub_mod) try: sub_spec.loader.exec_module(sub_mod) - except Exception: - pass + except Exception as exc: + sys.modules.pop(sub_name, None) + if getattr(mod, py_file.stem, None) is sub_mod: + delattr(mod, py_file.stem) + print( + f"warning: failed to bootstrap {sub_name}: {exc!r}", + file=sys.stderr, + ) return mod +def _voyage_pricing_table() -> dict[str, float]: + try: + from hermes_lcm.command import _VOYAGE_USD_PER_MILLION_TOKENS + except Exception as exc: + print( + f"warning: canonical command.py pricing unavailable ({exc!r}); " + "using standalone fallback", + file=sys.stderr, + ) + return _FALLBACK_VOYAGE_USD_PER_MILLION_TOKENS + return _VOYAGE_USD_PER_MILLION_TOKENS + + def _open_store(db_path: Path, asset_root: Path): ts = sys.modules["hermes_lcm.trajectory_store"] conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True) @@ -118,101 +139,102 @@ def main() -> int: asset_root = args.asset_root or args.db.parent store = _open_store(args.db, asset_root) - provider = ts.create_trajectory_embedding_provider( - args.provider, args.model, timeout_seconds=args.timeout, for_backfill=True, - ) - rate = _VOYAGE_USD_PER_MILLION_TOKENS.get(args.model, 0.06) - args.ledger.parent.mkdir(parents=True, exist_ok=True) - started = time.perf_counter() - checkpoint_state = {"logged_50pct": False} - - def _projected(stats: dict[str, Any]) -> tuple[float, float]: - cost = stats["billed_tokens"] / 1e6 * rate - embedded = max(1, stats["states_embedded"]) - pending = max(1, stats["pending"]) - projected_tokens = stats["billed_tokens"] / embedded * pending - projected_cost = projected_tokens / 1e6 * rate - return cost, projected_cost + try: + provider = ts.create_trajectory_embedding_provider( + args.provider, args.model, timeout_seconds=args.timeout, for_backfill=True, + ) + rate = _voyage_pricing_table().get(args.model, 0.06) + args.ledger.parent.mkdir(parents=True, exist_ok=True) + started = time.perf_counter() + checkpoint_state = {"logged_50pct": False} + + def _projected(stats: dict[str, Any]) -> tuple[float, float]: + cost = stats["billed_tokens"] / 1e6 * rate + embedded = max(1, stats["states_embedded"]) + pending = max(1, stats["pending"]) + projected_tokens = stats["billed_tokens"] / embedded * pending + projected_cost = projected_tokens / 1e6 * rate + return cost, projected_cost + + class _AbortCostCap(RuntimeError): + pass - class _AbortCostCap(RuntimeError): - pass + def _callback(stats: dict[str, Any]) -> None: + cost, projected_cost = _projected(stats) + elapsed = time.perf_counter() - started + record = { + "ts": time.time(), + "db": str(args.db), + "model": args.model, + "states_embedded": stats["states_embedded"], + "pending": stats["pending"], + "chunked_states": stats["chunked_states"], + "provider_calls": stats["provider_calls"], + "billed_tokens": stats["billed_tokens"], + "cost_usd": round(cost, 6), + "projected_total_usd": round(projected_cost, 6), + "elapsed_s": round(elapsed, 1), + } + with args.ledger.open("a") as handle: + handle.write(json.dumps(record) + "\n") + done = stats["states_embedded"] + pending = stats["pending"] + if pending and not checkpoint_state["logged_50pct"] and done >= pending / 2: + checkpoint_state["logged_50pct"] = True + print(f"[CHECKPOINT 50%] {done}/{pending} states, spent ${cost:.4f}, " + f"projected total ${projected_cost:.4f} (cap ${args.cost_cap})", + flush=True) + if projected_cost > args.cost_cap: + raise _AbortCostCap( + f"projected total ${projected_cost:.2f} exceeds cap ${args.cost_cap:.2f} " + f"after {done} states (${cost:.4f} spent)" + ) + if stats["provider_calls"] % 25 == 0: + print(f" {done}/{pending} embedded, {stats['provider_calls']} calls, " + f"${cost:.4f} spent, ~${projected_cost:.4f} projected " + f"({elapsed:.0f}s)", flush=True) + + print(f"== state backfill: {args.db} (model={args.model}, rate=${rate}/M) ==", + flush=True) + try: + stats = store.build_state_semantic_index( + provider, + resume=not args.no_resume, + batch_max_items=args.batch_items, + progress_callback=_callback, + ) + except _AbortCostCap as exc: + print(f"ABORTED (cost cap): {exc}", flush=True) + return 2 - def _callback(stats: dict[str, Any]) -> None: - cost, projected_cost = _projected(stats) elapsed = time.perf_counter() - started - record = { - "ts": time.time(), + cost = stats["billed_tokens"] / 1e6 * rate + summary = { "db": str(args.db), + "provider": args.provider, "model": args.model, - "states_embedded": stats["states_embedded"], - "pending": stats["pending"], + "rate_usd_per_million": rate, + "profile_digest": stats["profile_digest"], + "dim": stats["dim"], + "total_states": stats["total_states"], + "already_embedded_at_start": stats["already_embedded"], + "pending_at_start": stats["pending"], + "states_embedded_this_run": stats["states_embedded"], "chunked_states": stats["chunked_states"], "provider_calls": stats["provider_calls"], "billed_tokens": stats["billed_tokens"], "cost_usd": round(cost, 6), - "projected_total_usd": round(projected_cost, 6), - "elapsed_s": round(elapsed, 1), + "runtime_s": round(elapsed, 1), + "status": stats["status"], } - with args.ledger.open("a") as handle: - handle.write(json.dumps(record) + "\n") - done = stats["states_embedded"] - pending = stats["pending"] - if pending and not checkpoint_state["logged_50pct"] and done >= pending / 2: - checkpoint_state["logged_50pct"] = True - print(f"[CHECKPOINT 50%] {done}/{pending} states, spent ${cost:.4f}, " - f"projected total ${projected_cost:.4f} (cap ${args.cost_cap})", - flush=True) - if projected_cost > args.cost_cap: - raise _AbortCostCap( - f"projected total ${projected_cost:.2f} exceeds cap ${args.cost_cap:.2f} " - f"after {done} states (${cost:.4f} spent)" - ) - if stats["provider_calls"] % 25 == 0: - print(f" {done}/{pending} embedded, {stats['provider_calls']} calls, " - f"${cost:.4f} spent, ~${projected_cost:.4f} projected " - f"({elapsed:.0f}s)", flush=True) - - print(f"== state backfill: {args.db} (model={args.model}, rate=${rate}/M) ==", - flush=True) - try: - stats = store.build_state_semantic_index( - provider, - resume=not args.no_resume, - batch_max_items=args.batch_items, - progress_callback=_callback, - ) - except _AbortCostCap as exc: - print(f"ABORTED (cost cap): {exc}", flush=True) + if args.summary is not None: + args.summary.parent.mkdir(parents=True, exist_ok=True) + args.summary.write_text(json.dumps(summary, indent=2)) + print("== DONE ==", flush=True) + print(json.dumps(summary, indent=2), flush=True) + return 0 + finally: store.close() - return 2 - - elapsed = time.perf_counter() - started - cost = stats["billed_tokens"] / 1e6 * rate - summary = { - "db": str(args.db), - "provider": args.provider, - "model": args.model, - "rate_usd_per_million": rate, - "profile_digest": stats["profile_digest"], - "dim": stats["dim"], - "total_states": stats["total_states"], - "already_embedded_at_start": stats["already_embedded"], - "pending_at_start": stats["pending"], - "states_embedded_this_run": stats["states_embedded"], - "chunked_states": stats["chunked_states"], - "provider_calls": stats["provider_calls"], - "billed_tokens": stats["billed_tokens"], - "cost_usd": round(cost, 6), - "runtime_s": round(elapsed, 1), - "status": stats["status"], - } - store.close() - if args.summary is not None: - args.summary.parent.mkdir(parents=True, exist_ok=True) - args.summary.write_text(json.dumps(summary, indent=2)) - print("== DONE ==", flush=True) - print(json.dumps(summary, indent=2), flush=True) - return 0 if __name__ == "__main__": From f5d893bd31e3fdbf33efebb246ce2c051021ec5f Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 11:27:13 +0700 Subject: [PATCH 47/54] tests+stress: reconcile CI with the #168/#174 retrieval contract MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 5 test_lcm_engine grep/expand tests updated: they encoded pre-sanitization FTS behavior (adjudicated STALE per REGRESSION-REPORT method; each change annotated with the design change it tracks) - stress CLI: canary/secret containment checks now strip FTS5 snippet() match markers ('>>>'/'<<<') before matching — the planted token is present but was marker-split (>>>CANARY<<<_>>>SCOPE<<<...), which false-negatived recall canaries AND would have false-passed leak/secret detection. Applied at every containment site in both directions. - Known latent (not release-blocking): the stress smoke leaks process state that breaks test_lcm_engine when stress runs FIRST; CI's alphabetical order is unaffected (full suite verified). Tracked for follow-up. Adjudication: codex sol-high (lane A); stress fix + review: orchestrator. --- REGRESSION-REPORT.md | 51 +++++++++++++++++++++++++++++++++ benchmarking/stress.py | 34 +++++++++++++++------- tests/test_lcm_engine.py | 61 ++++++++++++++++++++++++++++++++-------- 3 files changed, 125 insertions(+), 21 deletions(-) create mode 100644 REGRESSION-REPORT.md diff --git a/REGRESSION-REPORT.md b/REGRESSION-REPORT.md new file mode 100644 index 000000000..045b0faf7 --- /dev/null +++ b/REGRESSION-REPORT.md @@ -0,0 +1,51 @@ +# Regression report — stress CLI canary false negatives + +## Affected test + +`tests/test_stress_release_check.py::test_stress_cli_smoke_writes_results_summary_and_uses_output_sandbox` + +## Verdict + +GENUINE REGRESSION. The stress CLI exits 1 even though `lcm_grep` returns the +requested canary rows. + +## Mechanism + +The #168 retrieval redesign keeps compound canary queries on FTS. FTS snippets +insert `>>>` and `<<<` around matched terms, splitting a planted token such as +`CANARY_SCOPE_A_000` into +`>>>CANARY<<<_>>>SCOPE<<<_>>>A<<<_>>>000<<<`. + +`benchmarking/stress.py` checks the serialized grep response for the original +contiguous token. The correct row is present, but the marker-split snippet makes +that string check false. The smoke run therefore reports: + +- `grep_canary_recall_miss` +- `all_scope_missing_cross_session_hit` +- `explicit_session_scope_missing_hit` + +This is a release-check CLI defect, not a retrieval miss and not a test +environment assumption. Per `SPEC.md`, product code is unchanged and the test +remains failing for the product-fix lane. + +## Minimal repro + +From worktree head `2edb8fc8e08cf533a4336e2d3d5c99d7772789b8`: + +```sh +PYTHONPATH=/Volumes/LEXAR/Codex/session-notes/2026-07-29/hermes-mono-pr-rounds/artifacts/agent-stub \ +/Volumes/LEXAR/Codex/session-notes/2026-07-29/hermes-mono-pr-rounds/artifacts/venv-ci-repro/bin/python \ +scripts/lcm_stress_check.py \ + --output /Volumes/LEXAR/Codex/session-notes/2026-07-29/hermes-mono-pr-rounds/artifacts/laneA-logs/stress-cli-repro \ + --tier smoke \ + --json +``` + +Observed: exit 1, `failure_count: 3`, correct grep rows present with marker-split +canary snippets, and empty stderr. + +## Evidence + +- `laneA-logs/stress-cli-stdout.log` +- `laneA-logs/stress-cli-stderr.log` +- `laneA-logs/stress-cli-repro/results/stress-results.json` diff --git a/benchmarking/stress.py b/benchmarking/stress.py index e9ae3085b..4ad588392 100644 --- a/benchmarking/stress.py +++ b/benchmarking/stress.py @@ -310,7 +310,7 @@ def deterministic_summary(*, text: str, source_tokens: int, token_budget: int, d def deterministic_expand_answer(*, prompt: str, context_blocks: list[dict[str, Any]], model: str, max_tokens: int, timeout: float) -> str: del prompt, model, max_tokens, timeout - serialized = json.dumps(context_blocks, ensure_ascii=False) + serialized = _fts_plain(json.dumps(context_blocks, ensure_ascii=False)) canaries = sorted(set(re.findall(r"CANARY_[A-Z0-9_]+\s*=\s*[A-Z0-9_:-]+", serialized))) return "Deterministic expansion answer. " + ("; ".join(canaries[:20]) if canaries else "No canaries found.") @@ -462,8 +462,22 @@ def _externalized_payload_files(hermes_home: str | Path) -> list[Path]: return sorted(payload_dir.glob("*.json")) +_FTS_SNIPPET_MARKERS = re.compile(r">>>|<<<") + + +def _fts_plain(serialized: str) -> str: + """Undo FTS5 snippet() match markers so containment checks see contiguous tokens. + + store.py renders grep snippets via snippet(messages_fts, 0, '>>>', '<<<', ...), + which splits a planted token like CANARY_SCOPE_A_000 into + >>>CANARY<<<_>>>SCOPE<<<_... — checks in BOTH directions (recall canaries AND + leak/secret detection) must match against the marker-free text. + """ + return _FTS_SNIPPET_MARKERS.sub("", serialized) + + def _json_contains(payload: Any, *needles: str) -> bool: - serialized = json.dumps(payload, ensure_ascii=False) + serialized = _fts_plain(json.dumps(payload, ensure_ascii=False)) return all(needle in serialized for needle in needles) @@ -559,7 +573,7 @@ def _case_multi_cycle_canary_recall(run: StressRun) -> None: for index in run.tier.multi_sample_indexes: cid = f"CANARY_LONG_{index:04d}" grep = run.call_tool(engine, "lcm_grep", {"query": cid, "limit": 5, "sort": "relevance"}) - hay = json.dumps(grep, ensure_ascii=False) + hay = _fts_plain(json.dumps(grep, ensure_ascii=False)) if cid not in hay or expected[cid] not in hay: missed.append({"canary": cid, "grep": grep}) continue @@ -574,7 +588,7 @@ def _case_multi_cycle_canary_recall(run: StressRun) -> None: for store_id in store_ids[:5]: expanded = run.call_tool(engine, "lcm_expand", {"store_id": store_id, "max_tokens": 500}) expanded_samples.append({"store_id": store_id, "expand": expanded}) - ehay = json.dumps(expanded, ensure_ascii=False) + ehay = _fts_plain(json.dumps(expanded, ensure_ascii=False)) if cid in ehay and expected[cid] in ehay: found_expanded = True break @@ -650,11 +664,11 @@ def _case_redaction_and_externalization_boundaries(run: StressRun) -> None: if leaked: run.fail(case, "sensitive_or_large_payload_leak", "Sensitive or oversized payload material was persisted raw across storage boundaries", {"leaked": leaked, "externalized_files": ext_files[:5]}) grep_secret = run.call_tool(engine, "lcm_grep", {"query": secret_values[0], "limit": 10}) - grep_secret_results_text = json.dumps(grep_secret.get("results", []), ensure_ascii=False) + grep_secret_results_text = _fts_plain(json.dumps(grep_secret.get("results", []), ensure_ascii=False)) if secret_values[0] in grep_secret_results_text: run.fail(case, "grep_returns_raw_secret", "lcm_grep returned a raw secret after sensitive-pattern redaction was enabled", {"grep": grep_secret}) grep_canary = run.call_tool(engine, "lcm_grep", {"query": "CANARY_SECRET_0001", "limit": 5}) - if "CANARY_SECRET_0001" not in json.dumps(grep_canary, ensure_ascii=False): + if "CANARY_SECRET_0001" not in _fts_plain(json.dumps(grep_canary, ensure_ascii=False)): run.fail(case, "redaction_broke_nonsecret_recall", "Sensitive redaction/externalization broke ordinary canary recall", {"grep": grep_canary}) run.record(case, "externalized_files", ext_files) run.record(case, "db_counts", _db_counts(db_path)) @@ -687,11 +701,11 @@ def _case_cross_session_scope_and_pagination(run: StressRun) -> None: cursor = load_a_1.get("next_cursor") or 0 load_a_2 = run.call_tool(engine, "lcm_load_session", {"session_id": "scope-a", "limit": 7, "after_store_id": cursor, "max_content_chars": 80}) - if "CANARY_SCOPE_A_000" in json.dumps(current_a.get("results", []), ensure_ascii=False): + if "CANARY_SCOPE_A_000" in _fts_plain(json.dumps(current_a.get("results", []), ensure_ascii=False)): run.fail(case, "current_scope_cross_session_leak", "lcm_grep current scope returned another session's raw content", {"current_result": current_a}) - if "CANARY_SCOPE_A_000" not in json.dumps(all_a.get("results", []), ensure_ascii=False): + if "CANARY_SCOPE_A_000" not in _fts_plain(json.dumps(all_a.get("results", []), ensure_ascii=False)): run.fail(case, "all_scope_missing_cross_session_hit", "lcm_grep session_scope=all failed to find another session's raw content", {"all_result": all_a}) - if "CANARY_SCOPE_A_000" not in json.dumps(explicit_a.get("results", []), ensure_ascii=False): + if "CANARY_SCOPE_A_000" not in _fts_plain(json.dumps(explicit_a.get("results", []), ensure_ascii=False)): run.fail(case, "explicit_session_scope_missing_hit", "lcm_grep session_scope=session failed to find the requested session content", {"explicit_result": explicit_a}) rows1 = load_a_1.get("messages") or load_a_1.get("rows") or [] rows2 = load_a_2.get("messages") or load_a_2.get("rows") or [] @@ -784,7 +798,7 @@ def reader(idx: int) -> None: if thread_errors: run.fail(case, "concurrent_read_write_errors", "Concurrent read/write smoke produced lock or internal errors", {"errors": thread_errors[:20]}) final = run.call_tool(engine, "lcm_grep", {"query": "CANARY_CONCURRENT_000", "limit": 5}) - if "CANARY_CONCURRENT_000" not in json.dumps(final, ensure_ascii=False): + if "CANARY_CONCURRENT_000" not in _fts_plain(json.dumps(final, ensure_ascii=False)): run.fail(case, "concurrent_old_canary_missing", "Old canary missing after concurrent read/write stress", {"grep": final}) run.record(case, "thread_errors_count", len(thread_errors)) run.record(case, "final_old_canary", final) diff --git a/tests/test_lcm_engine.py b/tests/test_lcm_engine.py index 048c95611..2acc3df5c 100644 --- a/tests/test_lcm_engine.py +++ b/tests/test_lcm_engine.py @@ -22236,7 +22236,7 @@ def test_handle_grep_like_fallback_recency_sorts_tied_cap_by_score(self, engine) assert result["results"][0]["store_id"] == best_id assert "best assistant" in result["results"][0]["snippet"] - def test_handle_grep_like_fallback_recency_sorts_tied_cap_by_directness(self, engine): + def test_handle_grep_like_fallback_recency_sorts_tied_cap_by_directness(self, engine, monkeypatch): best_id = engine._store.append( "test-session", {"role": "assistant", "content": "foo/bar baz direct assistant"}, @@ -22245,16 +22245,24 @@ def test_handle_grep_like_fallback_recency_sorts_tied_cap_by_directness(self, en "UPDATE messages SET timestamp = ? WHERE store_id = ?", (5000.0, best_id), ) + first_repeated_id = None for idx in range(600): store_id = engine._store.append( "test-session", {"role": "assistant", "content": f"foo/bar baz foo foo foo foo low assistant {idx}"}, ) + if first_repeated_id is None: + first_repeated_id = store_id engine._store._conn.execute( "UPDATE messages SET timestamp = ? WHERE store_id = ?", (5000.0, store_id), ) engine._store._conn.commit() + monkeypatch.setattr( + engine._store, + "_search_like", + lambda *args, **kwargs: pytest.fail("compound query fell back to LIKE"), + ) result = json.loads(engine.handle_tool_call( "lcm_grep", @@ -22264,8 +22272,9 @@ def test_handle_grep_like_fallback_recency_sorts_tied_cap_by_directness(self, en assert result["role"] == "assistant" assert len(result["results"]) == 1 assert result["results"][0]["role"] == "assistant" - assert result["results"][0]["store_id"] == best_id - assert "direct assistant" in result["results"][0]["snippet"] + # #168: compound punctuation is sanitized to terms and stays on FTS. + assert result["results"][0]["store_id"] == first_repeated_id + assert "low assistant 0" in result["results"][0]["snippet"] def test_handle_grep_like_fallback_recency_extends_tied_cap_for_json_penalty(self, engine): best_id = engine._store.append( @@ -23075,7 +23084,8 @@ def test_handle_grep_relevance_unmatched_quote_still_finds_results(self, engine) assert result["total_results"] == 1 assert result["results"][0]["type"] == "message" - assert result["results"][0]["snippet"].startswith("Keep vendoring out") + # #168: the unmatched quote is sanitized and the FTS hit keeps markers. + assert result["results"][0]["snippet"] == "Keep >>>vendoring<<< out of hermes-agent." def test_handle_grep_recency_same_timestamp_pool_matches_store_ordering(self, engine): engine._store.append_batch( @@ -23207,8 +23217,9 @@ def test_handle_grep_relevance_prefers_much_better_summary_over_vague_user_hit(s assert result["results"][0]["type"] == "summary" assert result["results"][0]["snippet"].startswith("Summary: keep hermes-lcm external") - assert result["results"][1]["type"] == "message" - assert result["results"][1]["role"] == "user" + # #168: the indexed conjunction excludes the vague partial message hit. + assert result["total_results"] == 1 + assert [item["type"] for item in result["results"]] == ["summary"] def test_handle_grep_hybrid_prefers_much_better_summary_over_vague_recent_user_hit(self, engine): store_id = engine._store.append( @@ -23239,8 +23250,9 @@ def test_handle_grep_hybrid_prefers_much_better_summary_over_vague_recent_user_h assert result["results"][0]["type"] == "summary" assert result["results"][0]["snippet"].startswith("Summary: keep hermes-lcm external") - assert result["results"][1]["type"] == "message" - assert result["results"][1]["role"] == "user" + # #168: hybrid inherits the FTS arm's conjunctive filtering. + assert result["total_results"] == 1 + assert [item["type"] for item in result["results"]] == ["summary"] def test_handle_grep_hybrid_does_not_let_weak_summary_beat_stronger_message_hit(self, engine): engine._store.append( @@ -25360,6 +25372,16 @@ def test_handle_expand_query_hyphenated_operator_query_falls_back_cleanly(self, lcm_tools, "_synthesize_expansion_answer", lambda **kwargs: "Recovered through normalized retrieval", ) + monkeypatch.setattr( + engine._store, + "_search_like", + lambda *args, **kwargs: pytest.fail("compound query fell back to LIKE"), + ) + monkeypatch.setattr( + engine._dag, + "_search_like", + lambda *args, **kwargs: pytest.fail("compound query fell back to LIKE"), + ) result = json.loads( engine.handle_tool_call( @@ -25372,9 +25394,26 @@ def test_handle_expand_query_hyphenated_operator_query_falls_back_cleanly(self, ) ) - assert result["answer"] == "Recovered through normalized retrieval" - assert result["node_ids"] == [node_id] - assert result["matches"] + # #168: raw bare OR tokens are literals, so the full conjunction has no match. + assert result["answer"] == "No matching summaries or raw messages found in the current session." + assert result["node_ids"] == [] + assert result["matches"] == [] + assert result["raw_matches"] == [] + + indexed = json.loads( + engine.handle_tool_call( + "lcm_expand_query", + { + "query": "plugin-only context-engine hermes-lcm stays external", + "prompt": "What were the agreements?", + "max_tokens": 500, + }, + ) + ) + + assert indexed["answer"] == "Recovered through normalized retrieval" + assert indexed["node_ids"] == [node_id] + assert indexed["matches"] def test_handle_expand_query_rejects_non_numeric_limits(self, engine): result = json.loads( From cfbfa904d6b55bf3c320758b750d2d0e55f28ac1 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 11:35:06 +0700 Subject: [PATCH 48/54] =?UTF-8?q?product:=20round-1=20review=20fixes=20?= =?UTF-8?q?=E2=80=94=2010=20confirmed=20defects=20+=20cheap=20hygiene=20(P?= =?UTF-8?q?R=20#175)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Validated-then-fixed (each with a focused red→green test): - trajectory: state-semantic arm now runs on lexical miss (additive tail, quota=0 byte-identical); profile activates only after backfill completes (interruption leaves staged rows, never an incomplete active index); provider failures fence to lexical + counted; query-embedding usage counted; dimension probe bounded; tiktoken-less chunking follows the shared estimator (fixes the chunked_states CI red — the ==1 expectation was independently wrong) - vector_store: batched full scan honors the caller's absolute deadline (default None = byte-identical; expiry → existing bounded-coverage disclosure) - recall: summary leads bounded by the clamped response limit and carry from_current_session for the expand-hint consumer; dead defer-reserve constant removed (orphaned by the derived-resume design) - evidence_compiler: mark_failed cleanup is best-effort, cannot replace the publish failure contract - + column-count hoist, annotation/docstring fixes, narrowed clock patch, direct DAG lineage tests, remediation apply coverage REFUTED (in FINDINGS-VERDICTS.md): MMR assert not load-bearing; state-id uniqueness guaranteed by PK+dedupe. DECLINED: SQLite row-value compat branch (no reproducible failure on supported matrix). Author: codex sol-high (lane B); review: orchestrator (cross-model). --- FINDINGS-VERDICTS.md | 19 ++ evidence_compiler.py | 7 +- retrieval_core.py | 2 + store.py | 5 +- tests/test_dag_source_message_ids.py | 61 +++++ tests/test_evidence_compiler.py | 40 ++++ tests/test_lcm_recall.py | 42 ++++ tests/test_schema_stamp_remediation.py | 8 + ...est_trajectory_state_semantic_expansion.py | 216 +++++++++++++++++- tests/test_vector_store.py | 49 +++- tools.py | 31 ++- trajectory_store.py | 118 ++++++++-- vector_store.py | 33 ++- 13 files changed, 579 insertions(+), 52 deletions(-) create mode 100644 FINDINGS-VERDICTS.md create mode 100644 tests/test_dag_source_message_ids.py diff --git a/FINDINGS-VERDICTS.md b/FINDINGS-VERDICTS.md new file mode 100644 index 000000000..18b3fa68f --- /dev/null +++ b/FINDINGS-VERDICTS.md @@ -0,0 +1,19 @@ +# Round-1 product findings verdicts + +| Item | Verdict | Source validation and disposition | Focused proof | +|---|---|---|---| +| 1 | CONFIRMED-FIXED | `state_semantic_quota > 0 and rows` excluded the empty-FTS underfill case. The state-semantic arm now runs for any positive quota and remains additive; quota zero is unchanged. | `test_expansion_can_fill_a_pure_lexical_miss`; existing default-byte and additive-tail tests. | +| 2 | CONFIRMED-FIXED | A profile was activated before any state vector write. Profiles now remain inactive throughout resumable writes, pass a complete-row-count check, and activate only at completion. An interrupted run leaves staged rows but no incomplete active profile; resume reuses them. | `test_profile_activates_only_after_resumed_backfill_completes`; existing idempotent/resume test. | +| 3 | CONFIRMED-FIXED | `_scan_ranked` checked only its relative budget. The absolute recall deadline now reaches summary and chunk KNN and is checked before and after each batch, producing bounded coverage on an early stop. | `test_full_scan_absolute_deadline_stops_between_batches`; existing budget/full-scan tests. | +| 4 | CONFIRMED-FIXED | `_semantic_state_ranks` provider exceptions escaped the query path. The state arm now fences them, increments the existing fallback counter, and leaves lexical results intact. | `test_state_query_provider_failure_degrades_to_lexical`. | +| 5 | CONFIRMED-FIXED | State query embeddings bypassed `_semantic_usage`. Query calls and successful usage tokens are now counted consistently with source-semantic queries. | `test_state_query_embedding_is_counted_in_semantic_usage`. | +| 6 | CONFIRMED-FIXED | The dimension probe sent the first full state document. It now probes only the first provider-bounded token chunk. | `test_dimension_probe_is_bounded_by_document_token_budget`. | +| 7 | CONFIRMED-FIXED | The fallback `token_budget * 4` window could exceed the shared estimator, especially for non-ASCII text; chunks now use that estimator directly. The reported `chunked_states == 1` test expectation was wrong independently: both fixture documents exceed five cl100k tokens, so the corrected contract is `2`. | `test_fallback_chunks_obey_the_shared_token_estimator`; corrected `test_chunked_path_pools_oversize_documents`. | +| 8 | CONFIRMED-FIXED | The hint consumer selects `lcm_expand(node_id=...)` only when `from_current_session` is present. Summary leads now preserve that context. | `test_summary_leads_preserve_current_context_and_obey_response_limit`. | +| 9 | CONFIRMED-FIXED | Summary leads followed the 50–100 row retrieval window while the response limit is at most 25 and may be lower. The actual clamped response limit now bounds leads. | `test_summary_leads_preserve_current_context_and_obey_response_limit`. | +| 10 | CONFIRMED-FIXED | `mark_failed` could raise inside the publish exception handler and replace the original failure contract. Cleanup is now best-effort and exception-safe. | `test_persist_compiled_view_cleanup_failure_does_not_escape`. | +| 11 | CONFIRMED-FIXED | Fixed the real/cheap items: state-cache freshness marker plus regression, removed the unused recall constant, corrected the summary-arm annotation, centralized the message column count, corrected the chunk-KNN docstring, narrowed the clock patch, added direct nested/source DAG tests, and extended query/trajectory remediation apply coverage. REFUTED: the MMR `assert` is not load-bearing because every nonempty `remaining` iteration assigns `best_item`; state-semantic IDs are unique by the embeddings PK, ranking indices, and merge dedupe. DECLINED: no SQLite row-value failure exists on the supported Python 3.11+ CI matrix, so no compatibility branch was added without a reproduced supported-path failure. | `test_state_matrix_cache_refreshes_after_same_profile_rewrite`; `tests/test_dag_source_message_ids.py`; extended schema/vector tests; touched-file suite. | + +V1 delivery flag: none. Items 1–7 remain default-off or change only an already-expired operation; item 8–11 changes do not alter default delivered-hit selection. + +Explicit exclusions were not changed: `store.py:1123` query semantics, arbitrary summary source-row selection, `tests/test_lcm_engine.py`, `test_stress_release_check.py`, and `benchmarking/`. diff --git a/evidence_compiler.py b/evidence_compiler.py index d7005eb33..ac9d769c8 100644 --- a/evidence_compiler.py +++ b/evidence_compiler.py @@ -1013,7 +1013,12 @@ def _persist_compiled_view(result: dict[str, Any], *, engine: Any) -> None: return except Exception as exc: if token is not None: - store.mark_failed(token, str(exc)) + try: + store.mark_failed(token, str(exc)) + except Exception: + # Cleanup is best-effort and must never replace the publish + # failure that selected this response contract. + pass result["persistence"].update( {"status": "error", "reason_code": "query_view_publish_failed"} ) diff --git a/retrieval_core.py b/retrieval_core.py index 90f9cb920..8175b832a 100644 --- a/retrieval_core.py +++ b/retrieval_core.py @@ -266,6 +266,7 @@ def run_knn( full_scan=full_scan, scan_max_rows=scan_max_rows, scan_budget_s=scan_budget_s, + deadline=deadline, ), ) @@ -314,6 +315,7 @@ def run_chunk_knn( full_scan=full_scan, scan_max_rows=scan_max_rows, scan_budget_s=scan_budget_s, + deadline=deadline, ), ) diff --git a/store.py b/store.py index 6cba71e47..b610e8745 100644 --- a/store.py +++ b/store.py @@ -60,6 +60,7 @@ "tool_calls, tool_name, timestamp, token_estimate, pinned, conversation_id, " "ingested_at, observed_at, observed_at_source" ) +_MESSAGE_SELECT_COLUMN_COUNT = len(_MESSAGE_SELECT_COLUMNS.split(",")) _UNKNOWN_SOURCE = "unknown" @@ -630,7 +631,7 @@ def scan_evidence_rows(self, *, limit: int = 4096) -> Dict[str, Any]: "truncated": False, "observed_at_missing_rows": 0, } - message_column_count = len(_MESSAGE_SELECT_COLUMNS.split(",")) + message_column_count = _MESSAGE_SELECT_COLUMN_COUNT total_rows = int(rows[0][message_column_count] or 0) snapshot_max_store_id = int(rows[0][message_column_count + 1] or 0) observed_at_missing_rows = int(rows[0][message_column_count + 2] or 0) @@ -1200,7 +1201,7 @@ def search(self, query: str, session_id: str | None = None, raw_primary_values: list[float] = [] for r in rows: d = self._row_to_dict(r) - base_columns = len(_MESSAGE_SELECT_COLUMNS.split(",")) + base_columns = _MESSAGE_SELECT_COLUMN_COUNT d["search_rank"] = r[base_columns] if len(r) > base_columns else None d["snippet"] = r[base_columns + 1] if len(r) > (base_columns + 1) else "" d["_directness_score"] = _message_directness_score(d.get("role"), d.get("content"), terms, phrases) diff --git a/tests/test_dag_source_message_ids.py b/tests/test_dag_source_message_ids.py new file mode 100644 index 000000000..8e5086775 --- /dev/null +++ b/tests/test_dag_source_message_ids.py @@ -0,0 +1,61 @@ +from hermes_lcm.config import LCMConfig +from hermes_lcm.dag import SummaryDAG, SummaryNode +from hermes_lcm.store import MessageStore + + +def _stores(tmp_path): + config = LCMConfig(database_path=str(tmp_path / "lcm.db")) + messages = MessageStore( + config.database_path, ingest_protection_config=config + ) + dag = SummaryDAG(config.database_path) + return messages, dag + + +def _message(messages, index: int) -> int: + return messages.append( + "session-a", {"role": "user", "content": f"message {index}"} + ) + + +def _node(dag, source_type: str, source_ids: list[int], depth: int) -> int: + return dag.add_node(SummaryNode( + session_id="session-a", + depth=depth, + summary=f"depth {depth}", + token_count=2, + source_token_count=4, + source_ids=source_ids, + source_type=source_type, + )) + + +def test_source_message_ids_deduplicates_orders_and_limits_direct_leaves(tmp_path): + messages, dag = _stores(tmp_path) + try: + first, second, third = [_message(messages, index) for index in range(3)] + node_id = _node( + dag, "messages", [third, first, third, second], depth=0 + ) + + assert dag.source_message_ids(node_id, limit=2) == [first, second] + assert dag.source_message_ids(node_id, limit=0) == [] + finally: + dag.close() + messages.close() + + +def test_source_message_ids_walks_nested_nodes_to_message_leaves(tmp_path): + messages, dag = _stores(tmp_path) + try: + first, second, third = [_message(messages, index) for index in range(3)] + left = _node(dag, "messages", [third, first], depth=0) + right = _node(dag, "messages", [second, third], depth=0) + parent = _node(dag, "nodes", [right, left], depth=1) + + assert dag.source_message_ids(parent, limit=10) == [ + first, second, third + ] + finally: + dag.close() + messages.close() diff --git a/tests/test_evidence_compiler.py b/tests/test_evidence_compiler.py index 1b6f668ac..7b6d0217f 100644 --- a/tests/test_evidence_compiler.py +++ b/tests/test_evidence_compiler.py @@ -850,6 +850,46 @@ def test_persist_compiled_view_releases_lease_on_publish_failure(tmp_path, monke assert retried_result["persistence"]["status"] == "published" +def test_persist_compiled_view_cleanup_failure_does_not_escape(tmp_path, monkeypatch): + engine = _engine(tmp_path) + query_views = QueryViewStore(engine._config.database_path) + engine._query_views = query_views + source = _append(engine, "Maya owns the Atlas rollout.") + selector = _selector( + _claim("owner", source, "atlas-owner", entity="Maya", role="user") + ) + try: + monkeypatch.setattr( + QueryViewStore, + "publish_ready", + lambda self, token, **kwargs: (_ for _ in ()).throw( + ValueError("original publish failure") + ), + ) + monkeypatch.setattr( + QueryViewStore, + "mark_failed", + lambda self, token, error: (_ for _ in ()).throw( + RuntimeError("cleanup failure") + ), + ) + + result = compile_evidence( + "Who owns the Atlas rollout?", + engine=engine, + baseline_refs=[source], + selector=selector, + enabled=True, + persist_view=True, + ) + finally: + query_views.close() + engine._store.close() + + assert result["persistence"]["status"] == "error" + assert result["persistence"]["reason_code"] == "query_view_publish_failed" + + def test_selective_persistence_rejects_generic_or_ungrounded_state(tmp_path): engine = _engine(tmp_path) query_views = QueryViewStore(engine._config.database_path) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index 2275a579e..69386f2e1 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -1812,6 +1812,48 @@ def test_summary_nodes_come_back_as_non_evidence_leads(recall_engine, monkeypatc assert "snippet" not in leads[0] +def test_summary_leads_preserve_current_context_and_obey_response_limit( + recall_engine, monkeypatch +): + node_ids = [] + for index, session_id in enumerate((CURRENT, "session-a", "session-b")): + source_id = recall_engine._store.append( + session_id, + {"role": "user", "content": f"budget note {index}"}, + source="chat", + ) + node_ids.append(_add_summary( + recall_engine, + f"kanban dashboard sprint rollup {index}", + session_id=session_id, + created_at=float(index + 1), + source_ids=[source_id], + )) + _seed_summary_vectors( + recall_engine, + [ + (node_ids[0], [1.0, 0.0]), + (node_ids[1], [0.0, 1.0]), + (node_ids[2], [0.0, 1.0]), + ], + ) + + payload = _recall( + recall_engine, + monkeypatch, + include="summaries", + detail="answer_ready", + scope_bias=0.0, + limit=1, + ) + + leads = payload["provenance"]["answer_ready"]["summary_leads"] + assert len(leads) == 1 + assert leads[0]["node_id"] == node_ids[0] + assert leads[0]["from_current_session"] is True + assert leads[0]["expand_hint"] == f"lcm_expand(node_id={node_ids[0]})" + + def test_admitted_hit_publishes_the_true_chunk_offset_not_zero( recall_engine, monkeypatch ): diff --git a/tests/test_schema_stamp_remediation.py b/tests/test_schema_stamp_remediation.py index c5f40fdf9..98e1b2f88 100644 --- a/tests/test_schema_stamp_remediation.py +++ b/tests/test_schema_stamp_remediation.py @@ -199,8 +199,16 @@ def test_classify_interim_stamp_with_query_view_and_trajectory_marker_tables(tmp conn = sqlite3.connect(db_path) try: assert classify_version_mismatch(conn) == db_bootstrap.VERSION_MISMATCH_INTERIM_STAMP + result = remediate_interim_schema_stamp(conn, apply=True) finally: conn.close() + assert result["status"] == "ok" + assert result["dropped_tables"] == [] + assert _stored_version(db_path) == db_bootstrap.SCHEMA_VERSION + assert { + "lcm_query_views", + "lcm_trajectory_corpora", + } <= _table_names(db_path) def test_classify_genuinely_newer_on_unknown_table(tmp_path): diff --git a/tests/test_trajectory_state_semantic_expansion.py b/tests/test_trajectory_state_semantic_expansion.py index 04c5795ba..f6d945e56 100644 --- a/tests/test_trajectory_state_semantic_expansion.py +++ b/tests/test_trajectory_state_semantic_expansion.py @@ -16,7 +16,11 @@ import hashlib from pathlib import Path +import struct +import pytest + +import hermes_lcm.tokens as token_module from hermes_lcm.trajectory_store import ( CorpusIdentity, TrajectorySource, @@ -40,6 +44,8 @@ class StateVectorProvider: def __init__(self) -> None: self.last_usage_tokens = 0 self.document_calls = 0 + self.query_calls = 0 + self.fail_queries = False @staticmethod def _vector(text: str) -> list[float]: @@ -56,10 +62,38 @@ def embed_documents(self, texts): return [self._vector(t) for t in texts] def embed_query(self, text): # noqa: ARG002 + self.query_calls += 1 + if self.fail_queries: + raise RuntimeError("simulated query provider failure") self.last_usage_tokens = 1 return [1.0, 0.0, 0.0] +class InterruptingStateVectorProvider(StateVectorProvider): + def __init__(self, *, model_id: str, fail_after: int | None) -> None: + super().__init__() + self.model_id = model_id + self.fail_after = fail_after + + def embed_documents(self, texts): + if self.fail_after is not None and self.document_calls >= self.fail_after: + raise RuntimeError("simulated interrupted backfill") + return super().embed_documents(texts) + + +class ProbeBudgetProvider(StateVectorProvider): + def __init__(self, token_limit: int) -> None: + super().__init__() + self.token_limit = token_limit + self.probe_documents: list[str] = [] + + def embed_query(self, text): + self.probe_documents.append(str(text)) + if token_module.count_tokens(str(text)) > self.token_limit: + raise ValueError("probe exceeded provider token limit") + return super().embed_query(text) + + def _identity() -> CorpusIdentity: return CorpusIdentity( dataset_name="example/state-semantic", @@ -188,6 +222,26 @@ def test_expansion_pulls_lexically_invisible_state_into_pool(tmp_path): assert by_state[answer]["rank"] == 1 # the alpha state is the top-ranked +def test_expansion_can_fill_a_pure_lexical_miss(tmp_path): + """The semantic tail exists for underfill, including an empty FTS pool.""" + store = _build_invisible_semantic_store(tmp_path) + baseline = store.query("lexically absent zephyr phrase", image_limit=0) + assert baseline == () + + expanded = store.query( + "lexically absent zephyr phrase", + image_limit=0, + include_adjacent=False, + state_semantic_quota=1, + ) + + assert len(expanded) == 1 + assert expanded[0].match_kind == "state_semantic" + assert _admitted(store)[0]["state_id"] == _state_id( + store, expanded[0].trajectory_id, expanded[0].state_index + ) + + def test_quota_caps_admissions(tmp_path): """Two lexically-invisible alpha states; a quota of 1 admits exactly one.""" asset_root = tmp_path / "assets" @@ -389,9 +443,105 @@ def test_backfill_is_idempotent_and_resumable(tmp_path): assert provider.document_calls == 0 +def test_profile_activates_only_after_resumed_backfill_completes(tmp_path): + asset_root = tmp_path / "assets" + asset_root.mkdir() + initial = StateVectorProvider() + store = TrajectoryStore( + tmp_path / "lcm.db", _identity(), asset_root=asset_root, + embedding_provider=initial, + ) + store.insert(_source( + asset_root, + trajectory_id="answerpath", + ordinal=0, + goal="Update the account settings", + texts=( + "widget configuration export panel form", + "alpha-answer success banner", + "logout footer copyright notice", + ), + )) + store.finalize(["answerpath"]) + store.build_state_semantic_index(initial) + + interrupted = InterruptingStateVectorProvider( + model_id="fake-state-v2", fail_after=1 + ) + with pytest.raises(RuntimeError, match="interrupted backfill"): + store.build_state_semantic_index( + interrupted, batch_max_items=1, batch_token_budget=100_000 + ) + + assert store.active_state_semantic_profile() is None + staged = store._conn.execute( + """ + SELECT profile_digest, active + FROM lcm_trajectory_state_embedding_profiles + WHERE model_name = ? + """, + ("fake-state-v2",), + ).fetchone() + assert staged is not None and int(staged["active"]) == 0 + staged_count = store._conn.execute( + "SELECT COUNT(*) FROM lcm_trajectory_state_embeddings " + "WHERE profile_digest = ?", + (staged["profile_digest"],), + ).fetchone()[0] + assert staged_count == 1 + + interrupted.fail_after = None + resumed = store.build_state_semantic_index( + interrupted, batch_max_items=1, batch_token_budget=100_000 + ) + active = store.active_state_semantic_profile() + assert resumed["already_embedded"] == 1 + assert resumed["states_embedded"] == 2 + assert active is not None + assert active["model_name"] == "fake-state-v2" + + +def test_dimension_probe_is_bounded_by_document_token_budget(tmp_path): + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = ProbeBudgetProvider(token_limit=5) + store = TrajectoryStore( + tmp_path / "lcm.db", _identity(), asset_root=asset_root, + embedding_provider=provider, + ) + store.insert(_source( + asset_root, + trajectory_id="answerpath", + ordinal=0, + goal="Update the account settings", + texts=("alpha-answer " + ("token " * 100),), + )) + store.finalize(["answerpath"]) + + stats = store.build_state_semantic_index( + provider, document_token_budget=5, batch_token_budget=50 + ) + + assert stats["states_embedded"] == 1 + assert provider.probe_documents + assert token_module.count_tokens(provider.probe_documents[0]) <= 5 + + +def test_fallback_chunks_obey_the_shared_token_estimator(tmp_path, monkeypatch): + monkeypatch.setattr(token_module, "_get_encoder", lambda: None) + store = object.__new__(TrajectoryStore) + document = "漢字かな交じり文" * 20 + + chunks = store._state_token_chunks(document, token_budget=5) + + assert "".join(chunks) == document + assert len(chunks) > 1 + assert all(token_module._fallback_token_estimate(chunk) <= 5 for chunk in chunks) + + def test_chunked_path_pools_oversize_documents(tmp_path): - """A document over the (test-lowered) per-document token budget takes the - chunked path and still yields exactly one usable per-state vector.""" + """Every document over the test-lowered token budget takes the chunk path + and each state still yields exactly one usable vector.""" asset_root = tmp_path / "assets" asset_root.mkdir() provider = StateVectorProvider() @@ -414,7 +564,9 @@ def test_chunked_path_pools_oversize_documents(tmp_path): stats = store.build_state_semantic_index( provider, document_token_budget=5, batch_token_budget=50, batch_max_items=4 ) - assert stats["chunked_states"] == 1 + # Both documents exceed five cl100k tokens; the old expectation of one + # chunked state confused word count with the provider tokenizer. + assert stats["chunked_states"] == 2 assert stats["states_embedded"] == 2 oversize = _state_id(store, "answerpath", 1) row = store._conn.execute( @@ -424,6 +576,64 @@ def test_chunked_path_pools_oversize_documents(tmp_path): assert row is not None and len(bytes(row["vector"])) == stats["dim"] * 4 +def test_state_query_provider_failure_degrades_to_lexical(tmp_path): + provider = StateVectorProvider() + store = _build_invisible_semantic_store(tmp_path, provider=provider) + baseline = store.query(_QUERY, image_limit=0) + fallbacks_before = store.semantic_metrics()["fallbacks"] + provider.fail_queries = True + + degraded = store.query( + _QUERY, image_limit=0, state_semantic_quota=8 + ) + + assert [hit.exact_ref for hit in degraded] == [ + hit.exact_ref for hit in baseline + ] + assert store.semantic_metrics()["fallbacks"] == fallbacks_before + 1 + + +def test_state_query_embedding_is_counted_in_semantic_usage(tmp_path): + provider = StateVectorProvider() + store = _build_invisible_semantic_store(tmp_path, provider=provider) + before = store.semantic_metrics() + + store.query(_QUERY, image_limit=0, state_semantic_quota=1) + + after = store.semantic_metrics() + assert after["query_calls"] == before["query_calls"] + 1 + assert after["query_tokens"] == before["query_tokens"] + 1 + + +def test_state_matrix_cache_refreshes_after_same_profile_rewrite(tmp_path): + provider = StateVectorProvider() + store = _build_invisible_semantic_store(tmp_path, provider=provider) + alpha = _state_id(store, "answerpath", 2) + beta = _state_id(store, "othertask", 0) + assert store._semantic_state_ranks("query", 1)[0][0] == alpha + + store._conn.execute( + """ + UPDATE lcm_trajectory_state_embeddings + SET vector = CASE state_id + WHEN ? THEN ? + WHEN ? THEN ? + ELSE vector + END, + embedded_at = embedded_at + 1000 + WHERE state_id IN (?, ?) + """, + ( + alpha, struct.pack("<3f", 0.0, 0.0, 1.0), + beta, struct.pack("<3f", 1.0, 0.0, 0.0), + alpha, beta, + ), + ) + store._conn.commit() + + assert store._semantic_state_ranks("query", 1)[0][0] == beta + + def test_arm_inert_without_provider_or_index(tmp_path): """Knob-on but no state index (or no provider) is a no-op, not an error.""" asset_root = tmp_path / "assets" diff --git a/tests/test_vector_store.py b/tests/test_vector_store.py index c92d7181a..a02852076 100644 --- a/tests/test_vector_store.py +++ b/tests/test_vector_store.py @@ -589,14 +589,51 @@ def test_full_scan_budget_stops_early_and_reports_bounded(tmp_path, monkeypatch) ) # A clock that advances a second per reading: the budget is spent after # the first batch, so the scan stops with 4 of 6 vectors unscored. - ticks = iter(range(1_000)) - monkeypatch.setattr( - vector_store_module.time, "monotonic", lambda: float(next(ticks)) - ) + with monkeypatch.context() as clock_patch: + ticks = iter(range(1_000)) + clock_patch.setattr( + vector_store_module.time, "monotonic", lambda: float(next(ticks)) + ) + result = store.knn( + [1.0, 0.0, 0.0], + k=1, + model="scan", + full_scan=True, + scan_budget_s=0.5, + ) - result = store.knn( - [1.0, 0.0, 0.0], k=1, model="scan", full_scan=True, scan_budget_s=0.5 + assert result.coverage == "bounded" + assert result.scanned == 2 + assert result.total == 6 + assert [row[0] for row in result] != [str(gold)] + finally: + store.close() + dag.close() + + +def test_full_scan_absolute_deadline_stops_between_batches(tmp_path, monkeypatch): + """The operation deadline remains a hard stop when the relative scan budget + is disabled (zero), so recall cannot start another full batch after expiry.""" + db_path = tmp_path / "full-scan-deadline.db" + dag = SummaryDAG(db_path) + store = VectorStore(db_path, bounded_scan_rows=2) + try: + gold = _seed_scan_corpus( + dag, store, 6, gold_vector=[1.0, 0.0, 0.0], filler_vector=[0.0, 1.0, 0.0] ) + with monkeypatch.context() as clock_patch: + ticks = iter((0.0, 0.0, 2.0)) + clock_patch.setattr( + vector_store_module.time, "monotonic", lambda: next(ticks) + ) + result = store.knn( + [1.0, 0.0, 0.0], + k=1, + model="scan", + full_scan=True, + scan_budget_s=0.0, + deadline=1.0, + ) assert result.coverage == "bounded" assert result.scanned == 2 diff --git a/tools.py b/tools.py index 560c6dbb1..b997d4700 100644 --- a/tools.py +++ b/tools.py @@ -294,10 +294,6 @@ def _parse_strict_int(value: Any, name: str) -> tuple[int | None, str | None]: # cursor in waves: a bounded number of batched reads per request rather than one # read per candidate it has to skip. _LCM_RECALL_STRICT_READ_WAVE = 32 -# How many density-blocked candidates the walk keeps in rank order awaiting a -# possible refund. Bounded because no more slots can ever be handed back than -# were admitted, so a reserve the size of the response cap is always enough. -_LCM_RECALL_STRICT_DEFER_RESERVE = _LCM_RECALL_LIMIT_CAP _LCM_RECALL_ANSWER_READY_CONTENT_CHARS = 2_400 # Recency boost half-life (30 days) and its floor: a memory's rank_score is # multiplied by 2**(-age/half_life), clamped so age never zeroes an otherwise @@ -4253,6 +4249,7 @@ def _lcm_recall_summary_source_hits( *, current: str | None, candidate_limit: int, + lead_limit: int, ) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: """Carry summary-KNN relevance onto the nodes' SOURCE MESSAGES. @@ -4281,11 +4278,17 @@ def _lcm_recall_summary_source_hits( ordered_ids: list[int] = [] seen: set[int] = set() for node, _score in nodes: - lead: dict[str, Any] = {"node_id": node.node_id, "session_id": node.session_id} - hint = _lcm_recall_summary_expand_hint({"session_id": node.session_id}) - if hint: - lead["expand_hint"] = hint - leads.append(lead) + if len(leads) < max(0, lead_limit): + lead: dict[str, Any] = { + "node_id": node.node_id, + "session_id": node.session_id, + "from_current_session": bool(current) + and node.session_id == current, + } + hint = _lcm_recall_summary_expand_hint(lead) + if hint: + lead["expand_hint"] = hint + leads.append(lead) if len(ordered_ids) >= candidate_limit: continue for store_id in engine._dag.source_message_ids( @@ -4331,9 +4334,10 @@ def _lcm_recall_summary_arm( query_vector: list[float], provider: Any, candidate_limit: int, + lead_limit: int, deadline: float, reference_strict: bool = False, -) -> tuple[list[dict[str, Any]], str]: +) -> tuple[list[dict[str, Any]], str, int | None, int | None, list[dict[str, Any]]]: """Summary KNN arm: embedded summaries across ALL sessions (no filter).""" knn_results = _run_within_deadline( lambda: run_knn( @@ -4370,7 +4374,11 @@ def _lcm_recall_summary_arm( current = engine.current_session_id if reference_strict: source_hits, leads = _lcm_recall_summary_source_hits( - engine, nodes, current=current, candidate_limit=candidate_limit + engine, + nodes, + current=current, + candidate_limit=candidate_limit, + lead_limit=lead_limit, ) return source_hits, coverage, knn_results.scanned, knn_results.total, leads hits: list[dict[str, Any]] = [] @@ -4699,6 +4707,7 @@ def lcm_recall(args: Dict[str, Any], **kwargs) -> str: query_vector=query_vector, provider=provider, candidate_limit=candidate_limit, + lead_limit=limit, deadline=deadline, reference_strict=reference_strict, ) diff --git a/trajectory_store.py b/trajectory_store.py index 163694a8e..70ec634c8 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -444,11 +444,14 @@ def __init__( self._last_semantic_attempt: TrajectorySemanticAttempt | None = None self._last_query_telemetry: dict[str, Any] | None = None # Lazily-populated per-STATE semantic matrix cache (issue #142): the - # tuple is ``(profile_digest, state_ids, matrix)`` where ``matrix`` is a - # normalized float32 array (numpy when available, else a list of tuples) - # so query-time state ranking is a single mat-vec instead of a per-row - # SQL + Python dot loop. Reset whenever a state backfill rewrites rows. - self._state_semantic_cache: tuple[str, list[int], Any] | None = None + # tuple is ``(profile_digest, freshness, state_ids, matrix)`` where + # ``freshness`` is the profile row count/latest-write marker and + # ``matrix`` is a normalized float32 array (numpy when available, else a + # list of tuples). The marker also catches supported same-profile + # rewrites that occur outside this instance's explicit cache reset. + self._state_semantic_cache: ( + tuple[str, tuple[int, float], list[int], Any] | None + ) = None self._lock = threading.RLock() self._conn = self._open_connection() try: @@ -1418,13 +1421,34 @@ def _state_token_chunks(self, document: str, token_budget: int) -> list[str]: cap). Uses the shared cl100k encoder -- the same tokenizer the provider's packing uses to gate the caps -- and falls back to a conservative character window if the encoder is unavailable.""" - from .tokens import _get_encoder + from .tokens import _fallback_token_estimate, _get_encoder encoder = _get_encoder() if encoder is None: - # ~4 chars/token is the module's own char fallback; stay under budget. - span = max(1, token_budget * 4) - return [document[i:i + span] for i in range(0, len(document), span)] + # Use the shared estimator itself rather than ``budget * 4``: + # the estimator includes a rounding token and deliberately assigns + # denser budgets to non-ASCII text, so a flat character window can + # exceed the same limit that selected this path. + chunks: list[str] = [] + start = 0 + while start < len(document): + low = 1 + high = min(len(document) - start, max(1, token_budget * 4)) + accepted = 0 + while low <= high: + middle = (low + high) // 2 + piece = document[start:start + middle] + if _fallback_token_estimate(piece) <= token_budget: + accepted = middle + low = middle + 1 + else: + high = middle - 1 + # A one-character piece always estimates to one token, but keep + # forward progress explicit if that estimator contract changes. + accepted = max(1, accepted) + chunks.append(document[start:start + accepted]) + start += accepted + return chunks or [document] token_ids = encoder.encode(document) chunks: list[str] = [] for start in range(0, len(token_ids), token_budget): @@ -1512,6 +1536,9 @@ def build_state_semantic_index( probe_doc = self._state_embed_document( all_states[0]["text"], all_states[0]["url"], int(all_states[0]["state_id"]) ) + probe_doc = self._state_token_chunks( + probe_doc, document_token_budget + )[0] probe_vector = _normalized_vector(active_provider.embed_query(probe_doc)) dim = len(probe_vector) @@ -1524,8 +1551,7 @@ def build_state_semantic_index( try: self._conn.execute( "UPDATE lcm_trajectory_state_embedding_profiles " - "SET active = 0 WHERE active = 1 AND profile_digest != ?", - (profile_digest,), + "SET active = 0 WHERE active = 1", ) self._conn.execute( """ @@ -1533,10 +1559,9 @@ def build_state_semantic_index( profile_digest, provider, model_name, dim, document_version, source_manifest_digest, state_count, active, created_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, 1, ?) + ) VALUES (?, ?, ?, ?, ?, ?, ?, 0, ?) ON CONFLICT(profile_digest) DO UPDATE SET - state_count = excluded.state_count, - active = 1 + state_count = excluded.state_count """, ( profile_digest, @@ -1683,6 +1708,33 @@ def _flush_normal() -> None: # A rewrite invalidates any cached query-time matrix. self._state_semantic_cache = None + embedded_count = int( + self._conn.execute( + "SELECT COUNT(*) FROM lcm_trajectory_state_embeddings " + "WHERE profile_digest = ?", + (profile_digest,), + ).fetchone()[0] + ) + if embedded_count != total_states: + raise TrajectoryStoreError( + "state semantic profile is incomplete after backfill" + ) + with self._lock: + self._conn.execute("BEGIN IMMEDIATE") + try: + self._conn.execute( + "UPDATE lcm_trajectory_state_embedding_profiles " + "SET active = 0 WHERE active = 1" + ) + self._conn.execute( + "UPDATE lcm_trajectory_state_embedding_profiles " + "SET active = 1 WHERE profile_digest = ?", + (profile_digest,), + ) + self._conn.commit() + except Exception: + self._conn.rollback() + raise stats["status"] = "current" if not pending else "built" return stats @@ -1869,9 +1921,22 @@ def _load_state_semantic_matrix( tuples for a pure-Python fallback -- the state vectors were normalized at backfill, so a query-vector dot product is cosine similarity either way. """ + freshness_row = self._conn.execute( + """ + SELECT COUNT(*), COALESCE(MAX(embedded_at), 0.0) + FROM lcm_trajectory_state_embeddings + WHERE profile_digest = ? + """, + (profile_digest,), + ).fetchone() + freshness = (int(freshness_row[0]), float(freshness_row[1])) cache = self._state_semantic_cache - if cache is not None and cache[0] == profile_digest: - return cache[1], cache[2] + if ( + cache is not None + and cache[0] == profile_digest + and cache[1] == freshness + ): + return cache[2], cache[3] rows = self._conn.execute( """ SELECT state_id, vector FROM lcm_trajectory_state_embeddings @@ -1892,7 +1957,9 @@ def _load_state_semantic_matrix( matrix = _np.zeros((0, int(dim)), dtype=" 0 and rows: + if state_semantic_quota > 0: pool_ids = {int(row["state_id"]) for row in rows} - ranked_states = self._semantic_state_ranks( - query, state_semantic_quota + len(pool_ids) + 16 - ) + try: + ranked_states = self._semantic_state_ranks( + query, state_semantic_quota + len(pool_ids) + 16 + ) + except Exception: + self._semantic_usage["fallbacks"] += 1 + ranked_states = [] score_by_state = {sid: score for sid, score in ranked_states} arm_semantic = [ {"state_id": sid} diff --git a/vector_store.py b/vector_store.py index 19ce60a1e..c3f454be0 100644 --- a/vector_store.py +++ b/vector_store.py @@ -1517,6 +1517,7 @@ def _scan_ranked( candidate_ids: Sequence[str], batch_rows: int, budget_s: float, + deadline: float | None, limit: int, score_batch: Any, ) -> tuple[list[tuple[str, float, str]], int, bool]: @@ -1528,9 +1529,10 @@ def _scan_ranked( the 25k recency window that made 86% of a 185k-vector corpus invisible to semantic recall (FINDING-F31 §2). - ``budget_s`` (0 = no early stop, the default) is the only thing that can - cut the scan short; when it does, the caller degrades to - ``coverage='bounded'`` and the existing disclosure names the ratio. + ``budget_s`` (0 = no relative early stop, the default) and the caller's + absolute operation ``deadline`` can cut the scan short; when either + does, the caller degrades to ``coverage='bounded'`` and the existing + disclosure names the ratio. Returns ``(ranked top-k, candidates scored, stopped early)``. A MULTI-BATCH sweep streams past the matrix LRU (``cache=False``). The @@ -1554,6 +1556,9 @@ def _scan_ranked( self._release_matrix_caches() started = time.monotonic() for start in range(0, len(candidate_ids), batch_rows): + if deadline is not None and time.monotonic() >= deadline: + stopped_early = True + break batch = candidate_ids[start:start + batch_rows] rowids, embedded_ids, kinds, scores = score_batch(batch, cache_batches) scanned += len(batch) @@ -1567,9 +1572,16 @@ def _scan_ranked( best.sort(key=self._rank_key) del best[limit:] exhausted = start + batch_rows >= len(candidate_ids) - if not exhausted and budget_s > 0 and (time.monotonic() - started) >= budget_s: - stopped_early = True - break + if not exhausted: + budget_expired = ( + budget_s > 0 and (time.monotonic() - started) >= budget_s + ) + deadline_expired = ( + deadline is not None and time.monotonic() >= deadline + ) + if budget_expired or deadline_expired: + stopped_early = True + break best.sort(key=self._rank_key) return ( [ @@ -1976,6 +1988,7 @@ def knn( full_scan: bool = False, scan_max_rows: int = 0, scan_budget_s: float = 0.0, + deadline: float | None = None, ) -> KNNResult: k = int(k) if k <= 0: @@ -2108,6 +2121,7 @@ def score_batch( candidate_ids=scan_ids, batch_rows=max(1, self.bounded_scan_rows), budget_s=scan_budget_s, + deadline=deadline, limit=k, score_batch=score_batch, ) @@ -2572,13 +2586,15 @@ def knn_chunks( full_scan: bool = False, scan_max_rows: int = 0, scan_budget_s: float = 0.0, + deadline: float | None = None, ) -> KNNResult: - """Bounded-candidate chunk KNN with the summary coverage contract. + """Chunk KNN with the summary coverage contract. Coverage is full|bounded|none exactly as for summaries: ``none`` when the corpus/identity is unbackfilled or a requested filter is unverifiable (missing message column), ``bounded`` when the scan was - cut short by a hard cap or latency budget, ``full`` otherwise. + cut short by a hard cap, latency budget, or operation deadline, and + ``full`` when the requested candidate set was scanned completely. """ k = int(k) if k <= 0: @@ -2689,6 +2705,7 @@ def score_batch( candidate_ids=scan_ids, batch_rows=max(1, self.bounded_scan_rows), budget_s=scan_budget_s, + deadline=deadline, limit=k, score_batch=score_batch, ) From f1eb364f386927ded1259efa0353da55c367f339 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 12:09:10 +0700 Subject: [PATCH 49/54] product+tooling: round-2 review fixes (PR #175) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - vector_store: absolute-deadline coverage completed on the int8/binary prescreen path (round-1 covered only the float batched scan); expiry → bounded coverage, unexpired ranking byte-identical - recall: summary lineage expansion is deadline-interruptible end-to-end (private read-only connection + progress-handler interrupt; TimeoutError degrades through the existing per-arm deadline machinery) - db_bootstrap: preserved families (lcm_query*/lcm_trajectory*) must verify EXACTLY before a stamp downgrade; unknown/partial/future shapes fail closed as genuinely-newer (verifiers now detect unexpected tables/columns) - trajectory: oversize chunks pack by item count AND cumulative token budget; dimension probe counted in backfill usage; state-arm fallback preserves typed failure reason - stress: one helper reads all supported lcm_grep result containers for every containment/scope check - backfill CLI: unknown model pricing errors loudly unless --assume-rate - docs: REGRESSION-REPORT paths repo-relative + final status; round-1 verdicts note the stress-CLI exception Deferred per train freeze (thread replied, tracked #172): sanitizer symbol preservation. Author: codex sol-high (R2 lane); review: orchestrator. --- FINDINGS-VERDICTS-R2.md | 19 ++ FINDINGS-VERDICTS.md | 5 +- REGRESSION-REPORT.md | 29 +-- benchmarking/state_embedding_backfill.py | 27 ++- benchmarking/stress.py | 34 ++- db_bootstrap.py | 27 +++ query_view_store.py | 15 ++ tests/test_int8_two_stage_knn.py | 42 ++++ tests/test_lcm_recall.py | 17 ++ tests/test_prescreen_flip_blackout.py | 33 +++ tests/test_schema_stamp_remediation.py | 113 +++++++++- tests/test_state_embedding_backfill_cli.py | 18 ++ tests/test_stress_release_check.py | 16 ++ ...est_trajectory_state_semantic_expansion.py | 62 +++++- tools.py | 99 ++++++--- trajectory_store.py | 177 +++++++++++++++- vector_store.py | 194 +++++++++++++++--- 17 files changed, 833 insertions(+), 94 deletions(-) create mode 100644 FINDINGS-VERDICTS-R2.md create mode 100644 tests/test_state_embedding_backfill_cli.py diff --git a/FINDINGS-VERDICTS-R2.md b/FINDINGS-VERDICTS-R2.md new file mode 100644 index 000000000..6b743ace7 --- /dev/null +++ b/FINDINGS-VERDICTS-R2.md @@ -0,0 +1,19 @@ +# Round-2 findings verdicts + +| Item / finding | Verdict | Source validation and disposition | Focused proof | +|---|---|---|---| +| 1 / `3671040912` | CONFIRMED-FIXED | Fully synced summary/chunk binary profiles bypassed all deadline checks. Binary loads now use deadline-interruptible read connections; Hamming batches, survivor loads, and rescoring check the same absolute deadline. Live two-stage output remains `full_approx`; actual expiry returns `bounded`. | `test_deadline_bounds_a_synced_binary_summary_prescreen`; `test_chunk_deadline_bounds_a_synced_binary_prescreen`; existing `test_two_stage_full_approx_coverage_surfaces_as_approximate`. | +| 2 / `3671040917` | CONFIRMED-FIXED | Oversize state chunks were sliced only by `batch_max_items`. Packing now stops on either item count or cumulative `batch_token_budget`, using the same estimator as normal documents. | `test_oversize_chunks_pack_by_item_and_token_budgets`; existing chunk/estimator tests. | +| 3 / `3671040939` | CONFIRMED-FIXED | `lcm_query*` and `lcm_trajectory*` tables passed the broad prefix allowlist but had no classifier verifier; remediation could lower the stamp while preserving an unknown shape. Current table/column/object shapes are now verified exactly, and malformed/future preserved families fail closed without drops or re-stamping. | Current full query/base+optional trajectory shapes classify interim; partial, extra-column, and unknown-table shapes classify genuinely newer and refuse apply. | +| 4 / `3671040926` | CONFIRMED-FIXED | Reference-strict summary lineage walked recursive source IDs and hydrated messages after the wrapped KNN/hydration stages, with no deadline. The expansion now uses an interruptible read-only snapshot and checks the deadline through lineage, hydration, and shaping. | `test_summary_source_expansion_refuses_an_expired_deadline`; recall suite. | +| 5 / `3671040922` | CONFIRMED-FIXED | The paid dimension probe ran before statistics initialization. Probe calls and `last_usage_tokens` now seed `provider_calls` and `billed_tokens`. | Extended `test_dimension_probe_is_bounded_by_document_token_budget`. | +| 6 / `3671050099` | CONFIRMED-FIXED | Stress containment/scope checks inconsistently read only `results` although other paths recognized `results`, `matches`, and `data`. One helper now supplies every such check. | `test_lcm_grep_result_rows_collects_every_supported_container`; stress smoke. | +| 7 / `3671050097` | CONFIRMED-FIXED | Unknown models silently used `$0.06/M`, weakening `--cost-cap`. Unknown pricing now errors before provider/store work unless `--assume-rate` explicitly supplies a positive rate. | `tests/test_state_embedding_backfill_cli.py`. | +| 8 / `3671050112` | CONFIRMED-FIXED | The state semantic exception arm incremented only the legacy fallback counter. It now records the typed exception/reason with the same fenced fallback telemetry as the source arm. | Extended `test_state_query_provider_failure_degrades_to_lexical`. | +| 9 / `3671050109` | CONFIRMED-FIXED | `REGRESSION-REPORT.md` embedded workstation paths and stale pre-fix status. The command now uses an explicit environment root plus repository-relative script/output paths and records final green status. | Absolute-path scan is empty; stress smoke passes. | +| 10 / `3671050103` | CONFIRMED-FIXED | The round-1 verdict document still said all `benchmarking/` work was excluded although the stress marker normalization landed. It now records that exception and passing smoke. | Documentation diff plus stress smoke. | +| 11 / `3671040934` | DEFERRED-NO-CHANGE | The symbol-loss claim reproduces in current semantics, but `SPEC.md` explicitly freezes this query-semantics change for this train and tracks it in fork issue `#172`. `search_query.py` and its tests were not changed. | Source inspection only; no current-lane test or code by triage. | + +V1 delivery flag: none. Unexpired prescreen ranking remains byte-identical and keeps `full_approx`; only actual deadline expiry changes coverage/content. The lineage change likewise stops only after expiry. Other fixes affect accounting, validation, diagnostics, benchmarks, or explicit CLI error handling rather than default V1 delivered-hit selection. + +Validation: touched area `226 passed`; full CI replica `2687 passed, 35 failed, 1 skipped, 12 xfailed`. Compared with `laneA-postfix-full-failures.txt`, there are zero new failure names and the prior oversize-chunk failure is resolved. Ruff and `git diff --check` pass. diff --git a/FINDINGS-VERDICTS.md b/FINDINGS-VERDICTS.md index 18b3fa68f..6f76e29bc 100644 --- a/FINDINGS-VERDICTS.md +++ b/FINDINGS-VERDICTS.md @@ -16,4 +16,7 @@ V1 delivery flag: none. Items 1–7 remain default-off or change only an already-expired operation; item 8–11 changes do not alter default delivered-hit selection. -Explicit exclusions were not changed: `store.py:1123` query semantics, arbitrary summary source-row selection, `tests/test_lcm_engine.py`, `test_stress_release_check.py`, and `benchmarking/`. +Round-1 follow-through: the stress-CLI FTS marker normalization in +`benchmarking/stress.py` landed with the round, and the final stress smoke passed. + +Explicit exclusions were not changed: `store.py:1123` query semantics, arbitrary summary source-row selection, and `tests/test_lcm_engine.py`; no benchmarking change beyond the documented stress-CLI fix belonged to round 1. diff --git a/REGRESSION-REPORT.md b/REGRESSION-REPORT.md index 045b0faf7..8a14a5c06 100644 --- a/REGRESSION-REPORT.md +++ b/REGRESSION-REPORT.md @@ -1,4 +1,4 @@ -# Regression report — stress CLI canary false negatives +# Regression report — stress CLI canary false negatives (fixed) ## Affected test @@ -6,8 +6,8 @@ ## Verdict -GENUINE REGRESSION. The stress CLI exits 1 even though `lcm_grep` returns the -requested canary rows. +GENUINE REGRESSION, FIXED. Before the fix, the stress CLI exited 1 even though +`lcm_grep` returned the requested canary rows. The final smoke test exits 0. ## Mechanism @@ -24,28 +24,33 @@ that string check false. The smoke run therefore reports: - `all_scope_missing_cross_session_hit` - `explicit_session_scope_missing_hit` -This is a release-check CLI defect, not a retrieval miss and not a test -environment assumption. Per `SPEC.md`, product code is unchanged and the test -remains failing for the product-fix lane. +This was a release-check CLI defect, not a retrieval miss and not a test +environment assumption. `benchmarking/stress.py` now strips the FTS markers +before every containment/scope assertion and reads all supported `lcm_grep` +result containers (`results`, `matches`, and `data`). ## Minimal repro -From worktree head `2edb8fc8e08cf533a4336e2d3d5c99d7772789b8`: +Set `HERMES_CI_REPRO_ROOT` to the local CI-replica artifact directory described +in the session-notes recipe, then run from the repository root: ```sh -PYTHONPATH=/Volumes/LEXAR/Codex/session-notes/2026-07-29/hermes-mono-pr-rounds/artifacts/agent-stub \ -/Volumes/LEXAR/Codex/session-notes/2026-07-29/hermes-mono-pr-rounds/artifacts/venv-ci-repro/bin/python \ +PYTHONPATH="$HERMES_CI_REPRO_ROOT/agent-stub" \ +"$HERMES_CI_REPRO_ROOT/venv-ci-repro/bin/python" \ scripts/lcm_stress_check.py \ - --output /Volumes/LEXAR/Codex/session-notes/2026-07-29/hermes-mono-pr-rounds/artifacts/laneA-logs/stress-cli-repro \ + --output .artifacts/stress-cli-repro \ --tier smoke \ --json ``` -Observed: exit 1, `failure_count: 3`, correct grep rows present with marker-split -canary snippets, and empty stderr. +Pre-fix: exit 1, `failure_count: 3`, correct grep rows present with marker-split +canary snippets, and empty stderr. Final: exit 0, `failure_count: 0`, and +`tests/test_stress_release_check.py::test_stress_cli_smoke_writes_results_summary_and_uses_output_sandbox` +passes. ## Evidence - `laneA-logs/stress-cli-stdout.log` - `laneA-logs/stress-cli-stderr.log` - `laneA-logs/stress-cli-repro/results/stress-results.json` +- `laneR2-logs/touched-area-final.xml` diff --git a/benchmarking/state_embedding_backfill.py b/benchmarking/state_embedding_backfill.py index 1dd6e5b73..52f440714 100644 --- a/benchmarking/state_embedding_backfill.py +++ b/benchmarking/state_embedding_backfill.py @@ -98,6 +98,21 @@ def _voyage_pricing_table() -> dict[str, float]: return _VOYAGE_USD_PER_MILLION_TOKENS +def _resolve_rate(model: str, assumed_rate: float | None) -> float: + if assumed_rate is not None: + rate = float(assumed_rate) + if rate <= 0: + raise ValueError("--assume-rate must be greater than zero") + return rate + pricing = _voyage_pricing_table() + if model not in pricing: + raise ValueError( + f"no pricing entry for model {model!r}; pass --assume-rate " + "USD-per-million-tokens to make the cost-cap assumption explicit" + ) + return float(pricing[model]) + + def _open_store(db_path: Path, asset_root: Path): ts = sys.modules["hermes_lcm.trajectory_store"] conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True) @@ -131,10 +146,21 @@ def main() -> int: parser.add_argument("--timeout", type=float, default=300.0) parser.add_argument("--cost-cap", type=float, default=10.0, help="abort if projected total cost exceeds this (USD)") + parser.add_argument( + "--assume-rate", + type=float, + default=None, + metavar="USD_PER_MILLION", + help="explicit USD-per-million-token rate for a model absent from pricing", + ) parser.add_argument("--no-resume", action="store_true") args = parser.parse_args() _bootstrap_package(_REPO_ROOT) + try: + rate = _resolve_rate(args.model, args.assume_rate) + except ValueError as exc: + parser.error(str(exc)) ts = sys.modules["hermes_lcm.trajectory_store"] asset_root = args.asset_root or args.db.parent store = _open_store(args.db, asset_root) @@ -143,7 +169,6 @@ def main() -> int: provider = ts.create_trajectory_embedding_provider( args.provider, args.model, timeout_seconds=args.timeout, for_backfill=True, ) - rate = _voyage_pricing_table().get(args.model, 0.06) args.ledger.parent.mkdir(parents=True, exist_ok=True) started = time.perf_counter() checkpoint_state = {"logged_50pct": False} diff --git a/benchmarking/stress.py b/benchmarking/stress.py index 4ad588392..e52784aa4 100644 --- a/benchmarking/stress.py +++ b/benchmarking/stress.py @@ -481,6 +481,18 @@ def _json_contains(payload: Any, *needles: str) -> bool: return all(needle in serialized for needle in needles) +def _lcm_grep_result_rows(payload: Any) -> list[Any]: + """Collect every result container supported by lcm_grep response variants.""" + if not isinstance(payload, dict): + return [] + rows: list[Any] = [] + for key in ("results", "matches", "data"): + container = payload.get(key) + if isinstance(container, list): + rows.extend(container) + return rows + + def _tool_recall_contains( run: StressRun, engine: Any, @@ -490,10 +502,11 @@ def _tool_recall_contains( ) -> tuple[bool, dict[str, Any]]: grep_args = {"query": query, "limit": 5, **(args or {})} grep = run.call_tool(engine, "lcm_grep", grep_args) - if _json_contains(grep, *needles): + grep_rows = _lcm_grep_result_rows(grep) + if _json_contains(grep_rows, *needles): return True, {"grep": grep, "expanded": []} expanded_samples: list[dict[str, Any]] = [] - for result in grep.get("results", []) + grep.get("matches", []) + grep.get("data", []): + for result in grep_rows: if not isinstance(result, dict) or not result.get("store_id"): continue expanded = run.call_tool(engine, "lcm_expand", {"store_id": int(result["store_id"]), "max_tokens": 800}) @@ -573,12 +586,13 @@ def _case_multi_cycle_canary_recall(run: StressRun) -> None: for index in run.tier.multi_sample_indexes: cid = f"CANARY_LONG_{index:04d}" grep = run.call_tool(engine, "lcm_grep", {"query": cid, "limit": 5, "sort": "relevance"}) - hay = _fts_plain(json.dumps(grep, ensure_ascii=False)) + grep_rows = _lcm_grep_result_rows(grep) + hay = _fts_plain(json.dumps(grep_rows, ensure_ascii=False)) if cid not in hay or expected[cid] not in hay: missed.append({"canary": cid, "grep": grep}) continue store_ids: list[int] = [] - for result in grep.get("results", []) + grep.get("matches", []) + grep.get("data", []): + for result in grep_rows: if isinstance(result, dict) and result.get("store_id"): store_ids.append(int(result["store_id"])) if not store_ids: @@ -664,7 +678,9 @@ def _case_redaction_and_externalization_boundaries(run: StressRun) -> None: if leaked: run.fail(case, "sensitive_or_large_payload_leak", "Sensitive or oversized payload material was persisted raw across storage boundaries", {"leaked": leaked, "externalized_files": ext_files[:5]}) grep_secret = run.call_tool(engine, "lcm_grep", {"query": secret_values[0], "limit": 10}) - grep_secret_results_text = _fts_plain(json.dumps(grep_secret.get("results", []), ensure_ascii=False)) + grep_secret_results_text = _fts_plain( + json.dumps(_lcm_grep_result_rows(grep_secret), ensure_ascii=False) + ) if secret_values[0] in grep_secret_results_text: run.fail(case, "grep_returns_raw_secret", "lcm_grep returned a raw secret after sensitive-pattern redaction was enabled", {"grep": grep_secret}) grep_canary = run.call_tool(engine, "lcm_grep", {"query": "CANARY_SECRET_0001", "limit": 5}) @@ -701,11 +717,13 @@ def _case_cross_session_scope_and_pagination(run: StressRun) -> None: cursor = load_a_1.get("next_cursor") or 0 load_a_2 = run.call_tool(engine, "lcm_load_session", {"session_id": "scope-a", "limit": 7, "after_store_id": cursor, "max_content_chars": 80}) - if "CANARY_SCOPE_A_000" in _fts_plain(json.dumps(current_a.get("results", []), ensure_ascii=False)): + if _json_contains(_lcm_grep_result_rows(current_a), "CANARY_SCOPE_A_000"): run.fail(case, "current_scope_cross_session_leak", "lcm_grep current scope returned another session's raw content", {"current_result": current_a}) - if "CANARY_SCOPE_A_000" not in _fts_plain(json.dumps(all_a.get("results", []), ensure_ascii=False)): + if not _json_contains(_lcm_grep_result_rows(all_a), "CANARY_SCOPE_A_000"): run.fail(case, "all_scope_missing_cross_session_hit", "lcm_grep session_scope=all failed to find another session's raw content", {"all_result": all_a}) - if "CANARY_SCOPE_A_000" not in _fts_plain(json.dumps(explicit_a.get("results", []), ensure_ascii=False)): + if not _json_contains( + _lcm_grep_result_rows(explicit_a), "CANARY_SCOPE_A_000" + ): run.fail(case, "explicit_session_scope_missing_hit", "lcm_grep session_scope=session failed to find the requested session content", {"explicit_result": explicit_a}) rows1 = load_a_1.get("messages") or load_a_1.get("rows") or [] rows2 = load_a_2.get("messages") or load_a_2.get("rows") or [] diff --git a/db_bootstrap.py b/db_bootstrap.py index 048946209..414f43339 100644 --- a/db_bootstrap.py +++ b/db_bootstrap.py @@ -292,6 +292,8 @@ def refuse_schema_version_too_new(conn: sqlite3.Connection) -> None: }, ) +_PRESERVED_FEATURE_FAMILIES = ("lcm_query", "lcm_trajectory") + def _family_verifier(prefix: str): """Return the final-shape verifier for a feature-family prefix, or ``None`` @@ -305,6 +307,14 @@ def _family_verifier(prefix: str): return verify_chunk_schema if prefix == "lcm_assertion": return verify_assertion_schema + if prefix == "lcm_query": + from .query_view_store import _verify_query_view_schema + + return _verify_query_view_schema + if prefix == "lcm_trajectory": + from .trajectory_store import _verify_trajectory_schema + + return _verify_trajectory_schema return None @@ -437,6 +447,23 @@ def classify_version_mismatch(conn: sqlite3.Connection) -> str: if _family_reports_newer_shape(findings, family_prefix=prefix): return VERSION_MISMATCH_GENUINELY_NEWER + # Query views and trajectories are preserved source/derived records, not + # disposable caches. Their init paths cannot safely rebuild an unknown or + # partial shape after this routine lowers the numeric stamp, so every + # present family must match this build exactly before downgrade. + for prefix in _PRESERVED_FEATURE_FAMILIES: + if not any(table.startswith(prefix) for table in tables): + continue + verify = _family_verifier(prefix) + if verify is None: + return VERSION_MISMATCH_GENUINELY_NEWER + try: + findings = verify(conn) + except (ImportError, sqlite3.DatabaseError): + return VERSION_MISMATCH_GENUINELY_NEWER + if findings: + return VERSION_MISMATCH_GENUINELY_NEWER + return VERSION_MISMATCH_INTERIM_STAMP diff --git a/query_view_store.py b/query_view_store.py index 29bd6cde4..a163b7ee3 100644 --- a/query_view_store.py +++ b/query_view_store.py @@ -470,6 +470,17 @@ def _ensure_query_view_schema(conn: sqlite3.Connection) -> None: def _verify_query_view_schema(conn: sqlite3.Connection) -> list[str]: missing: list[str] = [] + tables = { + str(row[0]) + for row in conn.execute( + "SELECT name FROM sqlite_master WHERE type = 'table' " + "AND name LIKE 'lcm_query_%'" + ) + } + missing.extend( + f"unexpected-table:{table}" + for table in sorted(tables - set(_REQUIRED_SCHEMA)) + ) for table, columns in _REQUIRED_SCHEMA.items(): actual = { str(row[1]) for row in conn.execute(f"PRAGMA table_info({table})") @@ -480,6 +491,10 @@ def _verify_query_view_schema(conn: sqlite3.Connection) -> list[str]: missing.extend( f"column:{table}.{column}" for column in sorted(columns - actual) ) + missing.extend( + f"unexpected-column:{table}.{column}" + for column in sorted(actual - columns) + ) required_objects = { "index": { "idx_lcm_query_view_sources_store", diff --git a/tests/test_int8_two_stage_knn.py b/tests/test_int8_two_stage_knn.py index aede400fd..3400a3482 100644 --- a/tests/test_int8_two_stage_knn.py +++ b/tests/test_int8_two_stage_knn.py @@ -144,6 +144,48 @@ def test_two_stage_reports_full_coverage(tmp_path): vs.close() +def test_chunk_deadline_bounds_a_synced_binary_prescreen(tmp_path): + db_path = tmp_path / "lcm.db" + _seed_messages(db_path, 2) + vs = VectorStore(db_path, bounded_scan_rows=1) + try: + i8 = _int8_identity(4) + vs.register_profile(MODEL, PROVIDER, 4, dtype="int8", task="chunk") + for idx, vec in enumerate( + ([1.0, 0.0, 0.0, 0.0], [0.0, 1.0, 0.0, 0.0]) + ): + vs.record_chunk_embedding( + f"{idx}:0", + MODEL, + vec, + store_id=idx, + chunk_index=0, + char_start=0, + char_end=1, + token_estimate=1, + identity=i8, + ) + + unbounded = vs.knn_chunks( + [1.0, 0.0, 0.0, 0.0], k=1, model=MODEL, provider=PROVIDER + ) + expired = vs.knn_chunks( + [1.0, 0.0, 0.0, 0.0], + k=1, + model=MODEL, + provider=PROVIDER, + full_scan=True, + deadline=-1.0, + ) + + assert unbounded.coverage == "full_approx" + assert expired.coverage == "bounded" + assert expired.scanned == 0 + assert expired.total == 2 + finally: + vs.close() + + # -- Stage-1 Hamming recall@M on a synthetic 5k set (the spec bar) ----------- diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index 69386f2e1..144120e42 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -1007,6 +1007,23 @@ def test_recall_query_timeout_has_its_own_budget(monkeypatch, tmp_path): assert cfg.embedding_query_timeout_s == 3.0 # grep's deadline untouched +def test_summary_source_expansion_refuses_an_expired_deadline(recall_engine): + node = SimpleNamespace( + node_id=1, + session_id="session-a", + ) + + with pytest.raises(TimeoutError, match="summary source expansion"): + lcm_tools._lcm_recall_summary_source_hits( + recall_engine, + [(node, 1.0)], + current=CURRENT, + candidate_limit=5, + lead_limit=1, + deadline=-1.0, + ) + + def test_recall_arm_weights_default_and_env_lenient(monkeypatch, tmp_path): """B2: recall_arm_weights default to fts=0.5,summary=1,chunk=1 and the env override parses leniently -- unknown arms, malformed pairs, and non-numeric diff --git a/tests/test_prescreen_flip_blackout.py b/tests/test_prescreen_flip_blackout.py index 462fa181f..034d187de 100644 --- a/tests/test_prescreen_flip_blackout.py +++ b/tests/test_prescreen_flip_blackout.py @@ -16,6 +16,7 @@ """ from __future__ import annotations +import hermes_lcm.vector_store as vector_store_module from hermes_lcm.config import LCMConfig from hermes_lcm.dag import SummaryDAG, SummaryNode from hermes_lcm.vector_store import VectorStore, _pack_sign_bits @@ -173,3 +174,35 @@ def test_scan_bounds_route_a_synced_binary_identity_to_the_exact_scan(tmp_path): assert budgeted.coverage in {"full", "bounded"} # exact path, never full_approx store.close() dag.close() + + +def test_deadline_bounds_a_synced_binary_summary_prescreen(tmp_path, monkeypatch): + """An expired deadline bounds the prescreen without changing its live path.""" + db_path = tmp_path / "deadline-prescreen.db" + dag = SummaryDAG(db_path) + store = VectorStore( + db_path, + config=LCMConfig(embedding_binary_prescreen=True), + bounded_scan_rows=1, + ) + store.register_profile(MODEL, PROVIDER, DIM) + _record(store, _add_summary(dag, created_at=1.0), [1.0, 0.0, 0.0]) + _record(store, _add_summary(dag, created_at=2.0), [0.0, 1.0, 0.0]) + + unbounded = store.knn([1.0, 0.0, 0.0], k=1, model=MODEL, full_scan=True) + ticks = iter((0.0, 0.0, 0.0, 2.0)) + monkeypatch.setattr( + vector_store_module.time, + "monotonic", + lambda: next(ticks, 2.0), + ) + expired = store.knn( + [1.0, 0.0, 0.0], k=1, model=MODEL, full_scan=True, deadline=1.0 + ) + + assert unbounded.coverage == "full_approx" + assert expired.coverage == "bounded" + assert expired.scanned == 1 + assert expired.total == 2 + store.close() + dag.close() diff --git a/tests/test_schema_stamp_remediation.py b/tests/test_schema_stamp_remediation.py index 98e1b2f88..14b9e0cac 100644 --- a/tests/test_schema_stamp_remediation.py +++ b/tests/test_schema_stamp_remediation.py @@ -26,8 +26,10 @@ remediate_interim_schema_stamp, ) from hermes_lcm.engine import LCMEngine +from hermes_lcm.query_view_store import QueryViewStore from hermes_lcm.rollup_store import RollupStore from hermes_lcm.store import MessageStore +from hermes_lcm.trajectory_store import CorpusIdentity, TrajectoryStore from hermes_lcm.vector_store import VectorStore @@ -119,6 +121,26 @@ def _add_early_chunk_tables(path: Path) -> None: conn.close() +def _add_current_query_and_trajectory_tables(path: Path, asset_root: Path) -> None: + query_views = QueryViewStore(path) + query_views.close() + trajectories = TrajectoryStore( + path, + CorpusIdentity( + dataset_name="schema-test", + dataset_revision="v1", + harness_commit="test", + tier="small", + domain="web", + ingest_config_digest="schema-test-v1", + ), + asset_root=asset_root, + ) + trajectories._ensure_semantic_schema() + trajectories._ensure_state_semantic_schema() + trajectories.close() + + def _table_names(path: Path) -> set[str]: conn = sqlite3.connect(path) try: @@ -177,11 +199,8 @@ def test_classify_interim_stamp_with_feature_marker_tables(tmp_path): conn.close() -def test_classify_interim_stamp_with_query_view_and_trajectory_marker_tables(tmp_path): - """lcm_query_*/lcm_trajectory_* are opt-in, marker-gated sidecars just like - rollup/embedding/chunk/assertion -- an interim stamp plus only these extra - tables must still classify as an interim stamp, not genuinely_newer - (F-PR436-3: the prefix whitelist previously stopped at lcm_assertion).""" +def test_classify_genuinely_newer_with_partial_query_and_trajectory_tables(tmp_path): + """Preserved families cannot be re-stamped from unverified marker tables.""" db_path = tmp_path / "lcm.db" _build_v5_db(db_path) conn = sqlite3.connect(db_path) @@ -198,19 +217,97 @@ def test_classify_interim_stamp_with_query_view_and_trajectory_marker_tables(tmp _stamp(db_path, db_bootstrap.SCHEMA_VERSION + 1) conn = sqlite3.connect(db_path) try: - assert classify_version_mismatch(conn) == db_bootstrap.VERSION_MISMATCH_INTERIM_STAMP + assert ( + classify_version_mismatch(conn) + == db_bootstrap.VERSION_MISMATCH_GENUINELY_NEWER + ) result = remediate_interim_schema_stamp(conn, apply=True) finally: conn.close() - assert result["status"] == "ok" + assert result["status"] == "refused" assert result["dropped_tables"] == [] - assert _stored_version(db_path) == db_bootstrap.SCHEMA_VERSION + assert _stored_version(db_path) == db_bootstrap.SCHEMA_VERSION + 1 assert { "lcm_query_views", "lcm_trajectory_corpora", } <= _table_names(db_path) +def test_classify_interim_stamp_with_current_query_and_trajectory_tables(tmp_path): + db_path = tmp_path / "lcm.db" + _build_v5_db(db_path) + asset_root = tmp_path / "assets" + asset_root.mkdir() + _add_current_query_and_trajectory_tables(db_path, asset_root) + _stamp(db_path, db_bootstrap.SCHEMA_VERSION + 1) + + conn = sqlite3.connect(db_path) + try: + assert ( + classify_version_mismatch(conn) + == db_bootstrap.VERSION_MISMATCH_INTERIM_STAMP + ) + result = remediate_interim_schema_stamp(conn, apply=True) + finally: + conn.close() + + assert result["status"] == "ok" + assert result["dropped_tables"] == [] + assert _stored_version(db_path) == db_bootstrap.SCHEMA_VERSION + + +@pytest.mark.parametrize( + ("table", "column"), + ( + ("lcm_query_views", "future_query_column"), + ("lcm_trajectory_corpora", "future_trajectory_column"), + ), +) +def test_classify_genuinely_newer_on_unverified_feature_column( + tmp_path, table, column +): + db_path = tmp_path / "lcm.db" + _build_v5_db(db_path) + conn = sqlite3.connect(db_path) + try: + conn.execute(f"CREATE TABLE {table} (id TEXT, {column} TEXT)") + conn.commit() + finally: + conn.close() + _stamp(db_path, db_bootstrap.SCHEMA_VERSION + 1) + conn = sqlite3.connect(db_path) + try: + assert ( + classify_version_mismatch(conn) + == db_bootstrap.VERSION_MISMATCH_GENUINELY_NEWER + ) + result = remediate_interim_schema_stamp(conn, apply=True) + finally: + conn.close() + assert result["status"] == "refused" + assert _stored_version(db_path) == db_bootstrap.SCHEMA_VERSION + 1 + + +def test_classify_genuinely_newer_on_unknown_query_family_table(tmp_path): + db_path = tmp_path / "lcm.db" + _build_v5_db(db_path) + conn = sqlite3.connect(db_path) + try: + conn.execute("CREATE TABLE lcm_query_future_widgets (id TEXT)") + conn.commit() + finally: + conn.close() + _stamp(db_path, db_bootstrap.SCHEMA_VERSION + 1) + conn = sqlite3.connect(db_path) + try: + assert ( + classify_version_mismatch(conn) + == db_bootstrap.VERSION_MISMATCH_GENUINELY_NEWER + ) + finally: + conn.close() + + def test_classify_genuinely_newer_on_unknown_table(tmp_path): db_path = tmp_path / "lcm.db" _build_v5_db(db_path) diff --git a/tests/test_state_embedding_backfill_cli.py b/tests/test_state_embedding_backfill_cli.py new file mode 100644 index 000000000..ef0fb1536 --- /dev/null +++ b/tests/test_state_embedding_backfill_cli.py @@ -0,0 +1,18 @@ +from __future__ import annotations + +import pytest + +from benchmarking import state_embedding_backfill + + +def test_resolve_rate_rejects_an_unknown_model_without_an_assumption(): + with pytest.raises(ValueError, match="no pricing entry"): + state_embedding_backfill._resolve_rate("voyage-future", None) + + +def test_resolve_rate_accepts_an_explicit_assumption_for_an_unknown_model(): + assert state_embedding_backfill._resolve_rate("voyage-future", 0.25) == 0.25 + + +def test_resolve_rate_keeps_canonical_rate_for_a_known_model(): + assert state_embedding_backfill._resolve_rate("voyage-4-large", None) == 0.12 diff --git a/tests/test_stress_release_check.py b/tests/test_stress_release_check.py index 8ba99b297..b36170472 100644 --- a/tests/test_stress_release_check.py +++ b/tests/test_stress_release_check.py @@ -63,6 +63,22 @@ def _assert_path_under(path_value: str, root: Path) -> None: assert Path(path_value).resolve().is_relative_to(root.resolve()) +def test_lcm_grep_result_rows_collects_every_supported_container(): + from benchmarking import stress + + payload = { + "results": [{"store_id": 1}], + "matches": [{"store_id": 2}], + "data": [{"store_id": 3}], + } + + assert stress._lcm_grep_result_rows(payload) == [ + {"store_id": 1}, + {"store_id": 2}, + {"store_id": 3}, + ] + + def test_stress_cli_query_fuzz_runs_without_hermes_agent_importable(tmp_path, monkeypatch): cli = _load_stress_cli() _block_agent_imports(monkeypatch) diff --git a/tests/test_trajectory_state_semantic_expansion.py b/tests/test_trajectory_state_semantic_expansion.py index f6d945e56..96465b46f 100644 --- a/tests/test_trajectory_state_semantic_expansion.py +++ b/tests/test_trajectory_state_semantic_expansion.py @@ -46,6 +46,7 @@ def __init__(self) -> None: self.document_calls = 0 self.query_calls = 0 self.fail_queries = False + self.usage_tokens_total = 0 @staticmethod def _vector(text: str) -> list[float]: @@ -59,6 +60,7 @@ def _vector(text: str) -> list[float]: def embed_documents(self, texts): self.document_calls += 1 self.last_usage_tokens = sum(max(1, len(str(t)) // 4) for t in texts) + self.usage_tokens_total += self.last_usage_tokens return [self._vector(t) for t in texts] def embed_query(self, text): # noqa: ARG002 @@ -66,6 +68,7 @@ def embed_query(self, text): # noqa: ARG002 if self.fail_queries: raise RuntimeError("simulated query provider failure") self.last_usage_tokens = 1 + self.usage_tokens_total += self.last_usage_tokens return [1.0, 0.0, 0.0] @@ -94,6 +97,20 @@ def embed_query(self, text): return super().embed_query(text) +class RequestBudgetProvider(StateVectorProvider): + def __init__(self, token_limit: int) -> None: + super().__init__() + self.token_limit = token_limit + self.request_token_counts: list[int] = [] + + def embed_documents(self, texts): + tokens = sum(token_module.count_tokens(str(text)) for text in texts) + self.request_token_counts.append(tokens) + if tokens > self.token_limit: + raise ValueError("request exceeded provider token limit") + return super().embed_documents(texts) + + def _identity() -> CorpusIdentity: return CorpusIdentity( dataset_name="example/state-semantic", @@ -525,6 +542,8 @@ def test_dimension_probe_is_bounded_by_document_token_budget(tmp_path): assert stats["states_embedded"] == 1 assert provider.probe_documents assert token_module.count_tokens(provider.probe_documents[0]) <= 5 + assert stats["provider_calls"] == provider.query_calls + provider.document_calls + assert stats["billed_tokens"] == provider.usage_tokens_total def test_fallback_chunks_obey_the_shared_token_estimator(tmp_path, monkeypatch): @@ -576,11 +595,45 @@ def test_chunked_path_pools_oversize_documents(tmp_path): assert row is not None and len(bytes(row["vector"])) == stats["dim"] * 4 +def test_oversize_chunks_pack_by_item_and_token_budgets(tmp_path): + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = RequestBudgetProvider(token_limit=9) + store = TrajectoryStore( + tmp_path / "lcm.db", + _identity(), + asset_root=asset_root, + embedding_provider=provider, + ) + store.insert( + _source( + asset_root, + trajectory_id="oversize", + ordinal=0, + goal="Exercise chunk packing", + texts=("alpha-answer " + ("token " * 100),), + ) + ) + store.finalize(["oversize"]) + + stats = store.build_state_semantic_index( + provider, + document_token_budget=5, + batch_token_budget=9, + batch_max_items=4, + ) + + assert stats["states_embedded"] == 1 + assert provider.request_token_counts + assert max(provider.request_token_counts) <= 9 + + def test_state_query_provider_failure_degrades_to_lexical(tmp_path): provider = StateVectorProvider() store = _build_invisible_semantic_store(tmp_path, provider=provider) baseline = store.query(_QUERY, image_limit=0) - fallbacks_before = store.semantic_metrics()["fallbacks"] + metrics_before = store.semantic_metrics() + attempts_before = store.semantic_attempt_counters() provider.fail_queries = True degraded = store.query( @@ -590,7 +643,12 @@ def test_state_query_provider_failure_degrades_to_lexical(tmp_path): assert [hit.exact_ref for hit in degraded] == [ hit.exact_ref for hit in baseline ] - assert store.semantic_metrics()["fallbacks"] == fallbacks_before + 1 + metrics_after = store.semantic_metrics() + attempts_after = store.semantic_attempt_counters() + assert metrics_after["fallbacks"] == metrics_before["fallbacks"] + 1 + assert attempts_after["fallbacks_by_reason"].get("other", 0) == ( + attempts_before["fallbacks_by_reason"].get("other", 0) + 1 + ) def test_state_query_embedding_is_counted_in_semantic_usage(tmp_path): diff --git a/tools.py b/tools.py index b997d4700..9258bdf21 100644 --- a/tools.py +++ b/tools.py @@ -4250,6 +4250,7 @@ def _lcm_recall_summary_source_hits( current: str | None, candidate_limit: int, lead_limit: int, + deadline: float, ) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: """Carry summary-KNN relevance onto the nodes' SOURCE MESSAGES. @@ -4274,35 +4275,71 @@ def _lcm_recall_summary_source_hits( budget. The purpose is to make the SESSION reachable with citable evidence, not to rank inside it. """ - leads: list[dict[str, Any]] = [] - ordered_ids: list[int] = [] - seen: set[int] = set() - for node, _score in nodes: - if len(leads) < max(0, lead_limit): - lead: dict[str, Any] = { - "node_id": node.node_id, - "session_id": node.session_id, - "from_current_session": bool(current) - and node.session_id == current, - } - hint = _lcm_recall_summary_expand_hint(lead) - if hint: - lead["expand_hint"] = hint - leads.append(lead) - if len(ordered_ids) >= candidate_limit: - continue - for store_id in engine._dag.source_message_ids( - node.node_id, limit=_LCM_RECALL_SUMMARY_SOURCE_PER_NODE - ): - if store_id in seen: + def require_remaining(stage: str) -> float: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError(f"summary source expansion deadline exhausted before {stage}") + return remaining + + require_remaining("database connection") + db_path = Path(engine._store.db_path).resolve() + uri = f"{db_path.as_uri()}?mode=ro" + conn: sqlite3.Connection | None = None + expired = [False] + + def interrupt_if_expired() -> int: + if time.monotonic() >= deadline: + expired[0] = True + return 1 + return 0 + + try: + conn = sqlite3.connect( + uri, + uri=True, + timeout=max(0.001, require_remaining("database connection")), + ) + conn.row_factory = sqlite3.Row + conn.execute("PRAGMA query_only=ON") + conn.set_progress_handler(interrupt_if_expired, 1000) + read_store = copy.copy(engine._store) + read_dag = copy.copy(engine._dag) + read_store._conn = conn + read_dag._conn = conn + read_dag._db_lock = threading.RLock() + + leads: list[dict[str, Any]] = [] + ordered_ids: list[int] = [] + seen: set[int] = set() + for node, _score in nodes: + require_remaining("lineage traversal") + if len(leads) < max(0, lead_limit): + lead: dict[str, Any] = { + "node_id": node.node_id, + "session_id": node.session_id, + "from_current_session": bool(current) + and node.session_id == current, + } + hint = _lcm_recall_summary_expand_hint(lead) + if hint: + lead["expand_hint"] = hint + leads.append(lead) + if len(ordered_ids) >= candidate_limit: continue - seen.add(store_id) - ordered_ids.append(store_id) + for store_id in read_dag.source_message_ids( + node.node_id, limit=_LCM_RECALL_SUMMARY_SOURCE_PER_NODE + ): + if store_id in seen: + continue + seen.add(store_id) + ordered_ids.append(store_id) + require_remaining("message hydration") - hits: list[dict[str, Any]] = [] - if ordered_ids: - rows = engine._store.get_batch(ordered_ids[:candidate_limit]) + hits: list[dict[str, Any]] = [] + if ordered_ids: + rows = read_store.get_batch(ordered_ids[:candidate_limit]) for store_id in ordered_ids[:candidate_limit]: + require_remaining("message shaping") row = rows.get(store_id) if row is None: continue @@ -4325,7 +4362,14 @@ def _lcm_recall_summary_source_hits( } hit["expand_hint"] = _lcm_recall_excerpt_expand_hint(hit) hits.append(hit) - return hits, leads + return hits, leads + except sqlite3.OperationalError as exc: + if expired[0] or time.monotonic() >= deadline: + raise TimeoutError("summary source expansion deadline exhausted") from exc + raise + finally: + if conn is not None: + conn.close() def _lcm_recall_summary_arm( @@ -4379,6 +4423,7 @@ def _lcm_recall_summary_arm( current=current, candidate_limit=candidate_limit, lead_limit=lead_limit, + deadline=deadline, ) return source_hits, coverage, knn_results.scanned, knn_results.total, leads hits: list[dict[str, Any]] = [] diff --git a/trajectory_store.py b/trajectory_store.py index 70ec634c8..ad9420c20 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -24,6 +24,7 @@ from .db_bootstrap import ( configure_connection, + get_fts_shadow_table_names, mark_migration_step_complete, refuse_schema_version_too_new, run_versioned_migrations, @@ -48,6 +49,142 @@ _STATE_EMBED_MAX_BATCH_ITEMS = 32 _STATE_EMBED_DOCUMENT_TOKEN_BUDGET = int(27_000 * 0.9) _STATE_EMBED_BATCH_TOKEN_BUDGET = int(80_000 * 0.9) + +_TRAJECTORY_BASE_SCHEMA: dict[str, frozenset[str]] = { + "lcm_trajectory_corpora": frozenset({ + "singleton", "identity_digest", "identity_json", "schema_version", + "corpus_uid", "haystack_digest", "source_manifest_digest", + "trajectory_count", "ingest_cursor", "status", "created_at", + "completed_at", + }), + "lcm_trajectory_sources": frozenset({ + "source_id", "trajectory_id", "ordinal", "source_json", "source_sha256", + "goal", "start_url", "outcome", "state_count", "inserted_at", + }), + "lcm_trajectory_states": frozenset({ + "state_id", "source_id", "state_index", "sequence_ordinal", "step", + "url", "incoming_action", "thoughts", "text", "search_text", + "observed_at", "observed_at_source", "occurred_at", + "occurred_at_source", "ingested_at", + }), + "lcm_trajectory_assets": frozenset({ + "asset_id", "state_id", "relative_path", "sha256", "byte_size", + }), + "lcm_trajectory_ingest_receipts": frozenset({ + "ordinal", "trajectory_id", "source_sha256", "committed_at", + }), + "lcm_trajectory_transitions": frozenset({ + "transition_id", "source_id", "sequence_ordinal", "pre_state_id", + "post_state_id", "incoming_action", + }), +} +_TRAJECTORY_OPTIONAL_SCHEMAS: tuple[dict[str, frozenset[str]], ...] = ( + { + "lcm_trajectory_embedding_profiles": frozenset({ + "profile_digest", "provider", "model_name", "dim", + "document_version", "source_manifest_digest", "document_count", + "index_digest", "active", "created_at", + }), + "lcm_trajectory_embeddings": frozenset({ + "source_id", "profile_digest", "document_sha256", "vector", + "embedded_at", + }), + }, + { + "lcm_trajectory_state_embedding_profiles": frozenset({ + "profile_digest", "provider", "model_name", "dim", + "document_version", "source_manifest_digest", "state_count", + "active", "created_at", + }), + "lcm_trajectory_state_embeddings": frozenset({ + "state_id", "profile_digest", "document_sha256", "vector", + "embedded_at", + }), + }, +) + + +def _verify_trajectory_schema(conn: sqlite3.Connection) -> list[str]: + """Return any non-current trajectory shape without mutating source data.""" + findings: list[str] = [] + tables = { + str(row[0]) + for row in conn.execute( + "SELECT name FROM sqlite_master WHERE type = 'table' " + "AND name LIKE 'lcm_trajectory%'" + ) + } + expected = dict(_TRAJECTORY_BASE_SCHEMA) + expected_fts = "lcm_trajectory_states_fts" + allowed = set(expected) | {expected_fts} + allowed.update(get_fts_shadow_table_names(expected_fts)) + for optional in _TRAJECTORY_OPTIONAL_SCHEMAS: + if tables.intersection(optional): + expected.update(optional) + allowed.update(optional) + + findings.extend( + f"unexpected-table:{table}" for table in sorted(tables - allowed) + ) + for table, columns in expected.items(): + actual = { + str(row[1]) for row in conn.execute(f"PRAGMA table_info({table})") + } + if not actual: + findings.append(f"table:{table}") + continue + findings.extend( + f"column:{table}.{column}" for column in sorted(columns - actual) + ) + findings.extend( + f"unexpected-column:{table}.{column}" + for column in sorted(actual - columns) + ) + if expected_fts not in tables: + findings.append(f"table:{expected_fts}") + + required_objects = { + "index": { + "lcm_trajectory_states_source_sequence", + *( + { + "lcm_trajectory_embedding_one_active", + "lcm_trajectory_embeddings_profile", + } + if "lcm_trajectory_embeddings" in expected + else set() + ), + *( + { + "lcm_trajectory_state_embedding_one_active", + "lcm_trajectory_state_embeddings_profile", + } + if "lcm_trajectory_state_embeddings" in expected + else set() + ), + }, + "trigger": { + "lcm_trajectory_fts_insert", + "lcm_trajectory_fts_delete", + "lcm_trajectory_fts_update", + }, + } + for object_type, names in required_objects.items(): + actual = { + str(row[0]) + for row in conn.execute( + "SELECT name FROM sqlite_master WHERE type = ? " + "AND name LIKE 'lcm_trajectory%'", + (object_type,), + ) + } + findings.extend( + f"{object_type}:{name}" for name in sorted(names - actual) + ) + findings.extend( + f"unexpected-{object_type}:{name}" for name in sorted(actual - names) + ) + return findings _MAX_CANDIDATES = 128 _MAX_RESULTS = 24 _MAX_IMAGES = 8 @@ -1514,6 +1651,8 @@ def build_state_semantic_index( # dim is authoritative; otherwise probe one state. source_profile = self._semantic_profile() dim: int | None = None + probe_provider_calls = 0 + probe_billed_tokens = 0 if ( source_profile is not None and str(source_profile["provider"]) == provider_name @@ -1541,6 +1680,10 @@ def build_state_semantic_index( )[0] probe_vector = _normalized_vector(active_provider.embed_query(probe_doc)) dim = len(probe_vector) + probe_provider_calls = 1 + probe_billed_tokens = max( + 0, int(getattr(active_provider, "last_usage_tokens", 0) or 0) + ) profile_digest = self._state_semantic_profile_digest( provider_name, model_name, dim, source_manifest_digest @@ -1599,8 +1742,8 @@ def build_state_semantic_index( "pending": len(pending), "states_embedded": 0, "chunked_states": 0, - "provider_calls": 0, - "billed_tokens": 0, + "provider_calls": probe_provider_calls, + "billed_tokens": probe_billed_tokens, } # Partition pending states into single-request documents and the @@ -1688,8 +1831,18 @@ def _flush_normal() -> None: for state_id, document, document_sha in oversize: chunks = self._state_token_chunks(document, document_token_budget) chunk_vectors: list[tuple[float, ...]] = [] - for start in range(0, len(chunks), batch_max_items): - sub = chunks[start:start + batch_max_items] + start = 0 + while start < len(chunks): + sub: list[str] = [] + sub_tokens = 0 + while start < len(chunks) and len(sub) < batch_max_items: + chunk = chunks[start] + chunk_tokens = count_tokens(chunk) + if sub and sub_tokens + chunk_tokens > batch_token_budget: + break + sub.append(chunk) + sub_tokens += chunk_tokens + start += 1 vectors = active_provider.embed_documents(sub) if len(vectors) != len(sub): raise ValueError("chunk embedding count does not match batch size") @@ -3234,12 +3387,26 @@ def query( state_semantic_admitted: list[dict[str, Any]] = [] if state_semantic_quota > 0: pool_ids = {int(row["state_id"]) for row in rows} + state_attempt_started = time.monotonic() try: ranked_states = self._semantic_state_ranks( query, state_semantic_quota + len(pool_ids) + 16 ) - except Exception: + except Exception as exc: self._semantic_usage["fallbacks"] += 1 + fallback_latency_ms = ( + time.monotonic() - state_attempt_started + ) * 1000.0 + try: + semantic_attempt = self._record_semantic_attempt( + outcome="fallback", + latency_ms=fallback_latency_ms, + exception=exc, + ) + except Exception: + semantic_attempt = self._record_minimal_fallback_attempt( + latency_ms=fallback_latency_ms, + ) ranked_states = [] score_by_state = {sid: score for sid, score in ranked_states} arm_semantic = [ diff --git a/vector_store.py b/vector_store.py index c3f454be0..8e0d047ea 100644 --- a/vector_store.py +++ b/vector_store.py @@ -7,6 +7,7 @@ from __future__ import annotations +import copy import hashlib import logging import math @@ -88,6 +89,12 @@ class _UnverifiableProvenance(RuntimeError): """A requested provenance filter could not be checked completely.""" +class _PrescreenDeadlineExpired(TimeoutError): + def __init__(self, scanned: int = 0) -> None: + super().__init__("binary prescreen deadline expired") + self.scanned = max(0, int(scanned)) + + def _require_supported_identity(identity: "EmbeddingIdentity") -> None: """Reject profile representations this implementation cannot encode.""" unsupported = [] @@ -1876,7 +1883,7 @@ def _load_embedding_binary_matrix( def _cached_binary_matrix( self, cache: "OrderedDict[tuple[str, int], tuple[list[str], Any]]", - identity_hash: str, loader, *, cacheable: bool, + identity_hash: str, loader, *, cacheable: bool, deadline: float | None, ) -> tuple[list[str], Any]: """Load a full-corpus binary matrix, caching only the UNFILTERED case. @@ -1885,20 +1892,71 @@ def _cached_binary_matrix( sign-bit matrix across calls (keyed on data_version for cross-process invalidation), which is what makes back-to-back two-stage recalls cheap. """ + if deadline is not None and time.monotonic() >= deadline: + raise _PrescreenDeadlineExpired() if not cacheable: - return loader() + loaded = loader() + if deadline is not None and time.monotonic() >= deadline: + raise _PrescreenDeadlineExpired() + return loaded with self._cache_lock: key = (identity_hash, self._data_version(identity_hash)) cached = cache.get(key) if cached is not None: cache.move_to_end(key) + if deadline is not None and time.monotonic() >= deadline: + raise _PrescreenDeadlineExpired() return cached loaded = loader() + if deadline is not None and time.monotonic() >= deadline: + raise _PrescreenDeadlineExpired() cache[key] = loaded while len(cache) > self._MATRIX_CACHE_MAX_ENTRIES: cache.popitem(last=False) return loaded + def _load_binary_with_deadline(self, loader, deadline: float | None): + """Run a full binary load on an interruptible read-only connection.""" + if deadline is None: + return loader(self) + remaining = deadline - time.monotonic() + if remaining <= 0: + raise _PrescreenDeadlineExpired() + uri = f"{self.db_path.resolve().as_uri()}?mode=ro" + conn: sqlite3.Connection | None = None + expired = [False] + + def interrupt_if_expired() -> int: + if time.monotonic() >= deadline: + expired[0] = True + return 1 + return 0 + + try: + conn = sqlite3.connect( + uri, + uri=True, + timeout=max(0.001, remaining), + check_same_thread=False, + ) + conn.row_factory = sqlite3.Row + conn.set_progress_handler(interrupt_if_expired, 1000) + reader = copy.copy(self) + reader._conn = conn + reader._write_lock = threading.RLock() + reader._cache_lock = threading.RLock() + loaded = loader(reader) + if time.monotonic() >= deadline: + raise _PrescreenDeadlineExpired() + return loaded + except sqlite3.OperationalError as exc: + if expired[0] or time.monotonic() >= deadline: + raise _PrescreenDeadlineExpired() from exc + raise + finally: + if conn is not None: + conn.close() + @staticmethod def _binary_rows_to_matrix(numpy: Any, rows: Sequence[sqlite3.Row]) -> tuple[list[str], Any]: ids: list[str] = [] @@ -1935,6 +1993,7 @@ def _two_stage_rank( binary_matrix: Any, *, chunk: bool, + deadline: float | None, ) -> list[tuple[str, float, str]]: """Stage-1 Hamming prescreen over ``binary_matrix`` then stage-2 rescore. @@ -1953,8 +2012,18 @@ def _two_stage_rank( # the dim so this should not happen). width = min(binary_matrix.shape[1], int(query_bits.shape[0])) popcount = numpy.asarray(_POPCOUNT_TABLE, dtype=numpy.uint16) - xor = numpy.bitwise_xor(binary_matrix[:, :width], query_bits[:width]) - hamming = popcount[xor].sum(axis=1) + hamming = numpy.empty(n, dtype=numpy.uint32) + prescreen_rows = min(max(1, self.bounded_scan_rows), 4096) + for start in range(0, n, prescreen_rows): + if deadline is not None and time.monotonic() >= deadline: + raise _PrescreenDeadlineExpired(start) + end = min(n, start + prescreen_rows) + xor = numpy.bitwise_xor( + binary_matrix[start:end, :width], query_bits[:width] + ) + hamming[start:end] = popcount[xor].sum(axis=1) + if deadline is not None and time.monotonic() >= deadline: + raise _PrescreenDeadlineExpired(end) m = min(n, max(k, self.knn_prescreen_multiplier * k)) if m < n: survivors = numpy.argpartition(hamming, m - 1)[:m] @@ -1966,13 +2035,19 @@ def _two_stage_rank( if chunk else self._load_vectors_for_ids ) + if deadline is not None and time.monotonic() >= deadline: + raise _PrescreenDeadlineExpired(n) rowids, out_ids, kinds, vectors = loader( identity_hash, dim, survivor_ids, dtype ) + if deadline is not None and time.monotonic() >= deadline: + raise _PrescreenDeadlineExpired(n) if not vectors: return [] matrix = numpy.asarray(vectors, dtype=numpy.float32) scores = matrix @ numpy.asarray(query, dtype=numpy.float32) + if deadline is not None and time.monotonic() >= deadline: + raise _PrescreenDeadlineExpired(n) return self._ranked(rowids, out_ids, kinds, scores, k) def knn( @@ -2035,21 +2110,50 @@ def knn( and self._binary_fully_synced(identity, chunk=False) ): unfiltered = since is None and until is None and conversation_ids is None - binary_ids, binary_matrix = self._cached_binary_matrix( - self._binary_matrix_cache, - identity, - lambda: self._load_embedding_binary_matrix( - numpy, identity, since=since, until=until, - conversation_ids=conversation_ids, - ), - cacheable=unfiltered, - ) + try: + binary_ids, binary_matrix = self._cached_binary_matrix( + self._binary_matrix_cache, + identity, + lambda: self._load_binary_with_deadline( + lambda reader: reader._load_embedding_binary_matrix( + numpy, + identity, + since=since, + until=until, + conversation_ids=conversation_ids, + ), + deadline, + ), + cacheable=unfiltered, + deadline=deadline, + ) + except _PrescreenDeadlineExpired as exc: + return KNNResult( + coverage="bounded", + scanned=exc.scanned, + total=self._count_embedded_vectors(identity, chunk=False), + ) if binary_matrix.shape[0] == 0: return KNNResult(coverage="none") - candidates = self._two_stage_rank( - numpy, identity, dim, dtype, query, k, binary_ids, binary_matrix, - chunk=False, - ) + try: + candidates = self._two_stage_rank( + numpy, + identity, + dim, + dtype, + query, + k, + binary_ids, + binary_matrix, + chunk=False, + deadline=deadline, + ) + except _PrescreenDeadlineExpired as exc: + return KNNResult( + coverage="bounded", + scanned=exc.scanned, + total=len(binary_ids), + ) # coverage='full_approx' (FIX 2): the whole corpus is REACHED, but # stage-1 Hamming keeps only M=mult*k survivors, so top-k is an # approximate (recall@M) result, not exact like the exact-scan 'full'. @@ -2640,21 +2744,51 @@ def knn_chunks( since is None and until is None and conversation_ids is None and source is None ) - binary_ids, binary_matrix = self._cached_binary_matrix( - self._chunk_binary_matrix_cache, - identity, - lambda: self._load_chunk_binary_matrix( - numpy, identity, since=since, until=until, - conversation_ids=conversation_ids, source=source, - ), - cacheable=unfiltered, - ) + try: + binary_ids, binary_matrix = self._cached_binary_matrix( + self._chunk_binary_matrix_cache, + identity, + lambda: self._load_binary_with_deadline( + lambda reader: reader._load_chunk_binary_matrix( + numpy, + identity, + since=since, + until=until, + conversation_ids=conversation_ids, + source=source, + ), + deadline, + ), + cacheable=unfiltered, + deadline=deadline, + ) + except _PrescreenDeadlineExpired as exc: + return KNNResult( + coverage="bounded", + scanned=exc.scanned, + total=self._count_embedded_vectors(identity, chunk=True), + ) if binary_matrix.shape[0] == 0: return KNNResult(coverage="none") - candidates = self._two_stage_rank( - numpy, identity, dim, dtype, query, k, binary_ids, binary_matrix, - chunk=True, - ) + try: + candidates = self._two_stage_rank( + numpy, + identity, + dim, + dtype, + query, + k, + binary_ids, + binary_matrix, + chunk=True, + deadline=deadline, + ) + except _PrescreenDeadlineExpired as exc: + return KNNResult( + coverage="bounded", + scanned=exc.scanned, + total=len(binary_ids), + ) # coverage='full_approx' (FIX 2): whole corpus reached, but stage-1 # keeps only M=mult*k survivors -> approximate top-k, not exact. return KNNResult(candidates, coverage="full_approx") From d9a1ab79d9a65470ed3501ff8b7eabaeed1d80cb Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 12:36:00 +0700 Subject: [PATCH 50/54] round-3 review fixes: close the final 6 (PR #175) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - query_view verifier discovers the full gated lcm_query% prefix — an unknown lcm_querycache table now classifies genuinely-newer and refuses downgrade (closes the bypass in round-2's preserved-family gate) - reference-strict delta: refs/progress/termination rebuilt from SURVIVING hits after 64K cap eviction — callers can no longer mark undelivered evidence as seen (cap-eviction path only; the bench single-shot path never engages the delta protocol) - backfill: cumulative progress emitted after EVERY chunk request — the CLI cost-cap and append-only spend ledger see mid-state spend; interruption preserves paid calls - rows always bound; shallow-copied reader gets its own write lock - _monotonic clock seam on deadline/budget paths; tests patch the seam, not process-wide stdlib time - chunk-side mid-prescreen deadline test (bounded coverage, scanned==1) Author: codex sol-high (R3); review: orchestrator. Full-suite exact-name parity vs baseline; 7 focused regressions green. --- FINDINGS-VERDICTS-R3.md | 21 +++++++++ query_view_store.py | 2 +- tests/test_int8_two_stage_knn.py | 24 ++++++++-- tests/test_lcm_recall.py | 46 +++++++++++++++++++ tests/test_prescreen_flip_blackout.py | 20 ++++++-- tests/test_schema_stamp_remediation.py | 5 +- ...est_trajectory_state_semantic_expansion.py | 45 ++++++++++++++++++ tests/test_vector_store.py | 4 +- tools.py | 23 +++++++++- trajectory_store.py | 1 + vector_store.py | 36 ++++++++------- 11 files changed, 197 insertions(+), 30 deletions(-) create mode 100644 FINDINGS-VERDICTS-R3.md diff --git a/FINDINGS-VERDICTS-R3.md b/FINDINGS-VERDICTS-R3.md new file mode 100644 index 000000000..af0128786 --- /dev/null +++ b/FINDINGS-VERDICTS-R3.md @@ -0,0 +1,21 @@ +# Round 3 findings verdicts + +Base: `fork/bench/w3b-on-wave1` at `4bd8401d96fb00fbcf37cc0ad0852a4ec7e0ea12`. + +1. `query_view_store.py` query-family discovery — **CONFIRMED-FIXED**. The verifier now discovers the full gated `lcm_query%` prefix. The `lcm_querycache` regression classifies the database as genuinely newer and refuses downgrade. +2. `tools.py` reference-strict delta after response-cap eviction — **CONFIRMED-FIXED**. Delta refs and progress fields are rebuilt from surviving delivered hits after whole-hit eviction. The regression asserts delivered refs equal delta refs and omitted refs remain unseen. **V1-DELIVERY-AFFECTING: cap-eviction path only.** +3. `trajectory_store.py` chunked-state spend progress — **CONFIRMED-FIXED**. Every successful chunk request emits cumulative progress before the next request. The low-cap regression stops after one chunk and preserves its provider-call and billed-token spend in the callback ledger. +4. `tools.py` conditional `rows` binding — **CONFIRMED-FIXED**. `rows` is always bound to the batch result or `{}`. The related shallow-copy nit is also aligned by replacing `read_store._write_lock`. +5. `vector_store.py` deadline clock seam — **CONFIRMED-FIXED**. `_monotonic` is module-local and used by deadline/budget paths; affected tests patch the seam. The summary mid-prescreen test is no longer coupled to process-wide time or an exact call sequence. +6. Chunk-side mid-prescreen deadline coverage — **CONFIRMED-FIXED**. The chunk test now expires after the first meaningful prescreen batch and asserts `coverage="bounded"` with `scanned == 1`; `bounded_scan_rows=1` now controls that exercised batch. + +Validation: + +- Exact regressions: 7 passed. +- Touched-area modules: 144 passed. +- Clock-seam delta: 2 passed. +- Full CI replica: 2689 passed, 35 failed, 1 skipped, 12 xfailed. +- Baseline comparison: 35 actual failure names exactly match 35 expected; zero new and zero missing. +- `git diff --check`: clean. + +No commit or push performed. diff --git a/query_view_store.py b/query_view_store.py index a163b7ee3..9ffb987f9 100644 --- a/query_view_store.py +++ b/query_view_store.py @@ -474,7 +474,7 @@ def _verify_query_view_schema(conn: sqlite3.Connection) -> list[str]: str(row[0]) for row in conn.execute( "SELECT name FROM sqlite_master WHERE type = 'table' " - "AND name LIKE 'lcm_query_%'" + "AND name LIKE 'lcm_query%'" ) } missing.extend( diff --git a/tests/test_int8_two_stage_knn.py b/tests/test_int8_two_stage_knn.py index 3400a3482..d18c20485 100644 --- a/tests/test_int8_two_stage_knn.py +++ b/tests/test_int8_two_stage_knn.py @@ -9,8 +9,10 @@ from __future__ import annotations import sqlite3 +import sys import numpy as np +import hermes_lcm.vector_store as vector_store_module from hermes_lcm.vector_store import ( EmbeddingIdentity, @@ -25,6 +27,17 @@ PROVIDER = "voyage" +def _expire_after_first_prescreen_batch() -> float: + frame = sys._getframe(1) + return ( + 2.0 + if frame.f_code.co_name == "_two_stage_rank" + and frame.f_locals.get("start") == 0 + and frame.f_locals.get("end") == 1 + else 0.0 + ) + + def _seed_messages(db_path, count, *, session="s", source="hist", ts0=0.0): conn = sqlite3.connect(str(db_path)) conn.execute( @@ -144,7 +157,7 @@ def test_two_stage_reports_full_coverage(tmp_path): vs.close() -def test_chunk_deadline_bounds_a_synced_binary_prescreen(tmp_path): +def test_chunk_deadline_bounds_a_synced_binary_prescreen(tmp_path, monkeypatch): db_path = tmp_path / "lcm.db" _seed_messages(db_path, 2) vs = VectorStore(db_path, bounded_scan_rows=1) @@ -169,18 +182,23 @@ def test_chunk_deadline_bounds_a_synced_binary_prescreen(tmp_path): unbounded = vs.knn_chunks( [1.0, 0.0, 0.0, 0.0], k=1, model=MODEL, provider=PROVIDER ) + monkeypatch.setattr( + vector_store_module, + "_monotonic", + _expire_after_first_prescreen_batch, + ) expired = vs.knn_chunks( [1.0, 0.0, 0.0, 0.0], k=1, model=MODEL, provider=PROVIDER, full_scan=True, - deadline=-1.0, + deadline=1.0, ) assert unbounded.coverage == "full_approx" assert expired.coverage == "bounded" - assert expired.scanned == 0 + assert expired.scanned == 1 assert expired.total == 2 finally: vs.close() diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index 144120e42..e9cf11d27 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -535,6 +535,52 @@ def test_answer_ready_delta_is_opt_in_and_returns_only_novel_exact_refs( assert exhausted["delta"]["termination_reason"] == "no_novel_exact_ref" +def test_answer_ready_delta_refs_match_hits_after_response_cap_eviction( + recall_engine, monkeypatch +): + contents = [ + f"kanban dashboard sprint evidence-{index} " + (str(index) * 2_300) + for index in range(3) + ] + store_ids = [ + recall_engine._store.append( + f"session-{index}", {"role": "user", "content": content} + ) + for index, content in enumerate(contents) + ] + all_refs = { + f"lcm:{store_id}:0-{len(content)}" + for store_id, content in zip(store_ids, contents) + } + monkeypatch.setattr(lcm_tools, "_LCM_RECALL_RESPONSE_CHAR_CAP", 6_000) + monkeypatch.setattr( + lcm_tools, + "_lcm_recall_summary_arm", + lambda *_a, **_k: ([], "none", 0, 0, []), + ) + monkeypatch.setattr( + lcm_tools, + "_lcm_recall_chunk_arm", + lambda *_a, **_k: ([], "none", 0, 0), + ) + + payload = _recall( + recall_engine, + monkeypatch, + include="verbatim", + detail="answer_ready", + limit=3, + seen_refs=[], + ) + + delivered_refs = [hit["exact_ref"] for hit in payload["hits"]] + assert payload["provenance"]["answer_ready"]["response_truncated"] is True + assert payload["delta"]["novel_refs"] == delivered_refs + assert payload["delta"]["novel_ref_count"] == len(delivered_refs) + assert all_refs - set(delivered_refs) + assert not (all_refs - set(delivered_refs)) & set(payload["delta"]["novel_refs"]) + + def test_answer_ready_baseline_bytes_ignore_disabled_occurrence_extension( recall_engine, monkeypatch ): diff --git a/tests/test_prescreen_flip_blackout.py b/tests/test_prescreen_flip_blackout.py index 034d187de..1d830d967 100644 --- a/tests/test_prescreen_flip_blackout.py +++ b/tests/test_prescreen_flip_blackout.py @@ -16,6 +16,8 @@ """ from __future__ import annotations +import sys + import hermes_lcm.vector_store as vector_store_module from hermes_lcm.config import LCMConfig from hermes_lcm.dag import SummaryDAG, SummaryNode @@ -27,6 +29,17 @@ DIM = 3 +def _expire_after_first_prescreen_batch() -> float: + frame = sys._getframe(1) + return ( + 2.0 + if frame.f_code.co_name == "_two_stage_rank" + and frame.f_locals.get("start") == 0 + and frame.f_locals.get("end") == 1 + else 0.0 + ) + + def _add_summary(dag: SummaryDAG, *, created_at: float) -> int: return dag.add_node( SummaryNode( @@ -190,11 +203,10 @@ def test_deadline_bounds_a_synced_binary_summary_prescreen(tmp_path, monkeypatch _record(store, _add_summary(dag, created_at=2.0), [0.0, 1.0, 0.0]) unbounded = store.knn([1.0, 0.0, 0.0], k=1, model=MODEL, full_scan=True) - ticks = iter((0.0, 0.0, 0.0, 2.0)) monkeypatch.setattr( - vector_store_module.time, - "monotonic", - lambda: next(ticks, 2.0), + vector_store_module, + "_monotonic", + _expire_after_first_prescreen_batch, ) expired = store.knn( [1.0, 0.0, 0.0], k=1, model=MODEL, full_scan=True, deadline=1.0 diff --git a/tests/test_schema_stamp_remediation.py b/tests/test_schema_stamp_remediation.py index 14b9e0cac..fa89b5fda 100644 --- a/tests/test_schema_stamp_remediation.py +++ b/tests/test_schema_stamp_remediation.py @@ -293,7 +293,7 @@ def test_classify_genuinely_newer_on_unknown_query_family_table(tmp_path): _build_v5_db(db_path) conn = sqlite3.connect(db_path) try: - conn.execute("CREATE TABLE lcm_query_future_widgets (id TEXT)") + conn.execute("CREATE TABLE lcm_querycache (id TEXT)") conn.commit() finally: conn.close() @@ -304,8 +304,11 @@ def test_classify_genuinely_newer_on_unknown_query_family_table(tmp_path): classify_version_mismatch(conn) == db_bootstrap.VERSION_MISMATCH_GENUINELY_NEWER ) + result = remediate_interim_schema_stamp(conn, apply=True) finally: conn.close() + assert result["status"] == "refused" + assert _stored_version(db_path) == db_bootstrap.SCHEMA_VERSION + 1 def test_classify_genuinely_newer_on_unknown_table(tmp_path): diff --git a/tests/test_trajectory_state_semantic_expansion.py b/tests/test_trajectory_state_semantic_expansion.py index 96465b46f..562ea75ac 100644 --- a/tests/test_trajectory_state_semantic_expansion.py +++ b/tests/test_trajectory_state_semantic_expansion.py @@ -628,6 +628,51 @@ def test_oversize_chunks_pack_by_item_and_token_budgets(tmp_path): assert max(provider.request_token_counts) <= 9 +def test_oversize_progress_can_stop_between_chunks_with_partial_spend(tmp_path): + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = StateVectorProvider() + store = TrajectoryStore( + tmp_path / "lcm.db", + _identity(), + asset_root=asset_root, + embedding_provider=provider, + ) + store.insert( + _source( + asset_root, + trajectory_id="oversize-cap", + ordinal=0, + goal="Trip the cap between chunk requests", + texts=("alpha-answer " + ("token " * 100),), + ) + ) + store.finalize(["oversize-cap"]) + ledger: list[dict] = [] + + class CostCapExceeded(RuntimeError): + pass + + def record_progress(stats): + ledger.append(dict(stats)) + if stats["billed_tokens"] > 1: + raise CostCapExceeded("low test cap exceeded") + + with pytest.raises(CostCapExceeded, match="low test cap"): + store.build_state_semantic_index( + provider, + document_token_budget=5, + batch_token_budget=5, + batch_max_items=1, + progress_callback=record_progress, + ) + + assert provider.document_calls == 1 + assert ledger[-1]["provider_calls"] == provider.query_calls + provider.document_calls + assert ledger[-1]["billed_tokens"] == provider.usage_tokens_total + assert ledger[-1]["states_embedded"] == 0 + + def test_state_query_provider_failure_degrades_to_lexical(tmp_path): provider = StateVectorProvider() store = _build_invisible_semantic_store(tmp_path, provider=provider) diff --git a/tests/test_vector_store.py b/tests/test_vector_store.py index a02852076..d38c35d9a 100644 --- a/tests/test_vector_store.py +++ b/tests/test_vector_store.py @@ -592,7 +592,7 @@ def test_full_scan_budget_stops_early_and_reports_bounded(tmp_path, monkeypatch) with monkeypatch.context() as clock_patch: ticks = iter(range(1_000)) clock_patch.setattr( - vector_store_module.time, "monotonic", lambda: float(next(ticks)) + vector_store_module, "_monotonic", lambda: float(next(ticks)) ) result = store.knn( [1.0, 0.0, 0.0], @@ -624,7 +624,7 @@ def test_full_scan_absolute_deadline_stops_between_batches(tmp_path, monkeypatch with monkeypatch.context() as clock_patch: ticks = iter((0.0, 0.0, 2.0)) clock_patch.setattr( - vector_store_module.time, "monotonic", lambda: next(ticks) + vector_store_module, "_monotonic", lambda: next(ticks) ) result = store.knn( [1.0, 0.0, 0.0], diff --git a/tools.py b/tools.py index 9258bdf21..cf3ed2aa3 100644 --- a/tools.py +++ b/tools.py @@ -4306,6 +4306,7 @@ def interrupt_if_expired() -> int: read_dag = copy.copy(engine._dag) read_store._conn = conn read_dag._conn = conn + read_store._write_lock = threading.RLock() read_dag._db_lock = threading.RLock() leads: list[dict[str, Any]] = [] @@ -4336,8 +4337,9 @@ def interrupt_if_expired() -> int: require_remaining("message hydration") hits: list[dict[str, Any]] = [] - if ordered_ids: - rows = read_store.get_batch(ordered_ids[:candidate_limit]) + rows = ( + read_store.get_batch(ordered_ids[:candidate_limit]) if ordered_ids else {} + ) for store_id in ordered_ids[:candidate_limit]: require_remaining("message shaping") row = rows.get(store_id) @@ -5192,6 +5194,23 @@ def _novel(entries: list[dict[str, Any]]) -> list[dict[str, Any]]: "content" in hit for hit in response["hits"] ) encoded = json.dumps(response, ensure_ascii=False) + if delta_requested: + novel_refs = [ + hit["exact_ref"] + for hit in response["hits"] + if hit.get("exact_ref") + ] + response["delta"].update( + { + "novel_ref_count": len(novel_refs), + "novel_refs": novel_refs, + "progress": bool(novel_refs), + "termination_reason": ( + None if novel_refs else "no_novel_exact_ref" + ), + } + ) + encoded = json.dumps(response, ensure_ascii=False) return encoded return json.dumps(response) diff --git a/trajectory_store.py b/trajectory_store.py index ad9420c20..754381e1e 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -1852,6 +1852,7 @@ def _flush_normal() -> None: stats["billed_tokens"] += max( 0, int(getattr(active_provider, "last_usage_tokens", 0) or 0) ) + _emit_progress() pooled = [sum(values) / len(chunk_vectors) for values in zip(*chunk_vectors)] normalized = _normalized_vector(pooled, expected_dim=dim) _persist([(state_id, document_sha, _pack_vector(normalized))]) diff --git a/vector_store.py b/vector_store.py index 8e0d047ea..89a710c01 100644 --- a/vector_store.py +++ b/vector_store.py @@ -39,6 +39,8 @@ logger = logging.getLogger(__name__) +_monotonic = time.monotonic + # Vectors are float32 in native little-endian order by default. These are # recorded as part of the canonical profile identity so a dtype/byteorder change # is detectable rather than silently reinterpreting stored bytes. @@ -1561,9 +1563,9 @@ def _scan_ranked( # store's whole lifetime. Release them before the first batch is # allocated, so peak really is one batch. self._release_matrix_caches() - started = time.monotonic() + started = _monotonic() for start in range(0, len(candidate_ids), batch_rows): - if deadline is not None and time.monotonic() >= deadline: + if deadline is not None and _monotonic() >= deadline: stopped_early = True break batch = candidate_ids[start:start + batch_rows] @@ -1581,10 +1583,10 @@ def _scan_ranked( exhausted = start + batch_rows >= len(candidate_ids) if not exhausted: budget_expired = ( - budget_s > 0 and (time.monotonic() - started) >= budget_s + budget_s > 0 and (_monotonic() - started) >= budget_s ) deadline_expired = ( - deadline is not None and time.monotonic() >= deadline + deadline is not None and _monotonic() >= deadline ) if budget_expired or deadline_expired: stopped_early = True @@ -1892,11 +1894,11 @@ def _cached_binary_matrix( sign-bit matrix across calls (keyed on data_version for cross-process invalidation), which is what makes back-to-back two-stage recalls cheap. """ - if deadline is not None and time.monotonic() >= deadline: + if deadline is not None and _monotonic() >= deadline: raise _PrescreenDeadlineExpired() if not cacheable: loaded = loader() - if deadline is not None and time.monotonic() >= deadline: + if deadline is not None and _monotonic() >= deadline: raise _PrescreenDeadlineExpired() return loaded with self._cache_lock: @@ -1904,11 +1906,11 @@ def _cached_binary_matrix( cached = cache.get(key) if cached is not None: cache.move_to_end(key) - if deadline is not None and time.monotonic() >= deadline: + if deadline is not None and _monotonic() >= deadline: raise _PrescreenDeadlineExpired() return cached loaded = loader() - if deadline is not None and time.monotonic() >= deadline: + if deadline is not None and _monotonic() >= deadline: raise _PrescreenDeadlineExpired() cache[key] = loaded while len(cache) > self._MATRIX_CACHE_MAX_ENTRIES: @@ -1919,7 +1921,7 @@ def _load_binary_with_deadline(self, loader, deadline: float | None): """Run a full binary load on an interruptible read-only connection.""" if deadline is None: return loader(self) - remaining = deadline - time.monotonic() + remaining = deadline - _monotonic() if remaining <= 0: raise _PrescreenDeadlineExpired() uri = f"{self.db_path.resolve().as_uri()}?mode=ro" @@ -1927,7 +1929,7 @@ def _load_binary_with_deadline(self, loader, deadline: float | None): expired = [False] def interrupt_if_expired() -> int: - if time.monotonic() >= deadline: + if _monotonic() >= deadline: expired[0] = True return 1 return 0 @@ -1946,11 +1948,11 @@ def interrupt_if_expired() -> int: reader._write_lock = threading.RLock() reader._cache_lock = threading.RLock() loaded = loader(reader) - if time.monotonic() >= deadline: + if _monotonic() >= deadline: raise _PrescreenDeadlineExpired() return loaded except sqlite3.OperationalError as exc: - if expired[0] or time.monotonic() >= deadline: + if expired[0] or _monotonic() >= deadline: raise _PrescreenDeadlineExpired() from exc raise finally: @@ -2015,14 +2017,14 @@ def _two_stage_rank( hamming = numpy.empty(n, dtype=numpy.uint32) prescreen_rows = min(max(1, self.bounded_scan_rows), 4096) for start in range(0, n, prescreen_rows): - if deadline is not None and time.monotonic() >= deadline: + if deadline is not None and _monotonic() >= deadline: raise _PrescreenDeadlineExpired(start) end = min(n, start + prescreen_rows) xor = numpy.bitwise_xor( binary_matrix[start:end, :width], query_bits[:width] ) hamming[start:end] = popcount[xor].sum(axis=1) - if deadline is not None and time.monotonic() >= deadline: + if deadline is not None and _monotonic() >= deadline: raise _PrescreenDeadlineExpired(end) m = min(n, max(k, self.knn_prescreen_multiplier * k)) if m < n: @@ -2035,18 +2037,18 @@ def _two_stage_rank( if chunk else self._load_vectors_for_ids ) - if deadline is not None and time.monotonic() >= deadline: + if deadline is not None and _monotonic() >= deadline: raise _PrescreenDeadlineExpired(n) rowids, out_ids, kinds, vectors = loader( identity_hash, dim, survivor_ids, dtype ) - if deadline is not None and time.monotonic() >= deadline: + if deadline is not None and _monotonic() >= deadline: raise _PrescreenDeadlineExpired(n) if not vectors: return [] matrix = numpy.asarray(vectors, dtype=numpy.float32) scores = matrix @ numpy.asarray(query, dtype=numpy.float32) - if deadline is not None and time.monotonic() >= deadline: + if deadline is not None and _monotonic() >= deadline: raise _PrescreenDeadlineExpired(n) return self._ranked(rowids, out_ids, kinds, scores, k) From 136112342749c6fb8cd83031e76ad08a1d1f27da Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 13:36:46 +0700 Subject: [PATCH 51/54] round-4 review fixes: all 8 findings from both reviewers (PR #175) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CodeRabbit batch: - strict-response cap now evicts summary leads after hits; delta refs stay consistent with survivors (cap-eviction path only) - deadline-bound binary loads on :memory:/memory-URI stores use the current interruptible connection instead of a filesystem RO URI - profile rebuild keeps the old complete profile SERVING until the new one cuts over atomically (removes the needless start-time deactivation — reviewer correctly showed the completion cutover already guarantees atomicity; availability restored during rebuilds) codex connector batch: - downgrade gate extended to schema OBJECTS: unknown lcm_query* triggers/ indexes classify genuinely-newer (tables-only was bypassable) - dimension-probe spend emitted to the ledger before any document request - recall_scan_budget_s is now a true operation-entry budget: candidate enumeration and matrix reads are interruptible (default 0.0 = disabled, V1 pins unaffected) - h3 replay aborts immediately on golden-gate failure (invalid-premise sweeps no longer produced) - temporal template terms match whole tokens ('update' no longer matches 'date'; sharp-compilation V2 path only) Author: codex sol-high (R4+R4B); review: orchestrator. Full-suite exact-name parity; 8 focused regressions green. --- FINDINGS-VERDICTS-R4.md | 17 ++++ FINDINGS-VERDICTS-R4B.md | 19 ++++ benchmarking/h3_composition_replay.py | 3 + query_view_store.py | 5 + tests/test_benchmarking_replay.py | 30 ++++++ tests/test_int8_two_stage_knn.py | 55 +++++++++++ tests/test_lcm_recall.py | 48 +++++++++ tests/test_schema_stamp_remediation.py | 46 +++++++++ tests/test_trajectory_compact_delivery.py | 11 +++ ...est_trajectory_state_semantic_expansion.py | 63 +++++++++++- tests/test_vector_store.py | 70 +++++++++++-- tools.py | 9 +- trajectory_store.py | 18 ++-- vector_store.py | 99 +++++++++++++++---- 14 files changed, 456 insertions(+), 37 deletions(-) create mode 100644 FINDINGS-VERDICTS-R4.md create mode 100644 FINDINGS-VERDICTS-R4B.md diff --git a/FINDINGS-VERDICTS-R4.md b/FINDINGS-VERDICTS-R4.md new file mode 100644 index 000000000..727f5d581 --- /dev/null +++ b/FINDINGS-VERDICTS-R4.md @@ -0,0 +1,17 @@ +# Round 4 findings verdicts + +Merge: `d9a1ab79d9a65470ed3501ff8b7eabaeed1d80cb` fast-forwarded to `fece680d5f016b2a22cefb69e7b634a88e4f70d2`. + +| Item | Verdict | Disposition | +|---|---|---| +| 1 strict response cap | CONFIRMED-FIXED | Evict hits, then summary leads; rebuild delta refs from surviving hits only. | +| 2 in-memory deadline load | CONFIRMED-FIXED | `:memory:` and memory-URI loads use the current interruptible connection. | +| 3 profile rebuild continuity | CONFIRMED-FIXED | Keep the old profile active until the completed replacement cuts over atomically. | + +V1-DELIVERY-AFFECTING: item 1 only, at cap eviction. +Focused regressions: 3 passed (`laneR4-logs/focused-r4.xml`). +Touched area: 181 passed (`laneR4-logs/touched-area-r4.xml`). +Ruff on changed source/tests: clean (`laneR4-logs/ruff-r4.txt`). +Full replica: 2691 passed, 35 failed, 1 skipped, 12 xfailed (`laneR4-logs/full-suite-r4.xml`). +Baseline comparison: 35 actual = 35 expected; zero new/missing failure names (`laneR4-logs/full-suite-r4-comparison.txt`). +`git diff --check`: clean. No commit or push performed. diff --git a/FINDINGS-VERDICTS-R4B.md b/FINDINGS-VERDICTS-R4B.md new file mode 100644 index 000000000..9e847594b --- /dev/null +++ b/FINDINGS-VERDICTS-R4B.md @@ -0,0 +1,19 @@ +# Round 4B findings verdicts + +PR `100yenadmin/hermes-lcm#175` @ `fece680d5f016b2a22cefb69e7b634a88e4f70d2`. +The spec's single-comment URL returned 404; all listed bodies were fetched by the canonical `pulls/comments/` route. +Comment `3671346839` is the already-fixed R4 profile-continuity item; the five R4B items follow. + +| Item | Verdict | Disposition | +|---|---|---| +| 1 query objects | CONFIRMED-FIXED | Reject extra `lcm_query*` indexes/triggers; both probes classify genuinely newer. Trajectory already rejects both extra object types. | +| 2 probe ledger | CONFIRMED-FIXED | Emit the probe's calls/tokens before any document request; callback can stop spend immediately. | +| 3 scan budget | CONFIRMED-FIXED | Config promises a hard scan budget; unlimited enumeration had no deadline. Start at operation entry and interrupt summary/chunk candidate reads. | +| 4 H3 golden gate | CONFIRMED-FIXED | Print failure and return 1 before ground truth, sweep, latency work, or artifact writing. | +| 5 temporal tokens | CONFIRMED-FIXED | Match temporal terms as whole tokens; `update` no longer matches `date`. | + +Item 5 affects only sharp-compilation V2 (`TrajectoryStore.query` with `sharp_token_budget > 0`); V1 message-store recall is untouched. +Focused modules: 125 passed (`laneR4B-logs/focused-modules-final.xml`); Ruff and `git diff --check`: clean. +Full replica: 2697 passed, 35 failed, 1 skipped, 12 xfailed (`laneR4B-logs/full-suite-r4b.xml`). +Baseline: exact same 35 R4 failure names; zero new/missing (`laneR4B-logs/full-suite-r4b-comparison.txt`). +No commit or push performed. diff --git a/benchmarking/h3_composition_replay.py b/benchmarking/h3_composition_replay.py index dbc14ef36..1d344a1c8 100644 --- a/benchmarking/h3_composition_replay.py +++ b/benchmarking/h3_composition_replay.py @@ -396,6 +396,9 @@ def main() -> int: golden = golden_gate(ctx) print(f" {golden['passed']}/{golden['total']} byte-identical; " f"failures={golden['failures'][:10]}", flush=True) + if golden["passed"] != golden["total"]: + print("GOLDEN GATE FAILED -- aborting before sweep", flush=True) + return 1 ground = load_ground_truth(ctx) ceiling = measure_ceiling(ctx, ground["recovery_targets"]) diff --git a/query_view_store.py b/query_view_store.py index 9ffb987f9..69ece2639 100644 --- a/query_view_store.py +++ b/query_view_store.py @@ -517,6 +517,11 @@ def _verify_query_view_schema(conn: sqlite3.Connection) -> list[str]: missing.extend( f"{object_type}:{name}" for name in sorted(names - actual) ) + gated = {name for name in actual if name.startswith("lcm_query")} + missing.extend( + f"unexpected-{object_type}:{name}" + for name in sorted(gated - names) + ) return missing diff --git a/tests/test_benchmarking_replay.py b/tests/test_benchmarking_replay.py index 64569023b..732c27bca 100644 --- a/tests/test_benchmarking_replay.py +++ b/tests/test_benchmarking_replay.py @@ -2,9 +2,11 @@ import json from pathlib import Path +import sys import pytest +import benchmarking.h3_composition_replay as h3_composition_replay import hermes_lcm.engine as lcm_engine from benchmarking.fixtures import make_synthetic_fixture @@ -45,6 +47,34 @@ def test_replay_below_threshold_does_not_compress(tmp_path): assert Path(metrics.database_path).is_relative_to(tmp_path) +def test_h3_composition_replay_aborts_when_golden_gate_fails( + tmp_path, monkeypatch, capsys +): + output = tmp_path / "invalid-sweep.json" + monkeypatch.setattr( + h3_composition_replay, "ReplayContext", lambda *args, **kwargs: object() + ) + monkeypatch.setattr( + h3_composition_replay, + "golden_gate", + lambda _ctx: {"passed": 0, "total": 1, "failures": ["q1"]}, + ) + monkeypatch.setattr( + h3_composition_replay, + "load_ground_truth", + lambda _ctx: pytest.fail("sweep continued after failed golden gate"), + ) + monkeypatch.setattr( + sys, + "argv", + ["h3_composition_replay.py", "--out", str(output)], + ) + + assert h3_composition_replay.main() == 1 + assert "GOLDEN GATE FAILED -- aborting before sweep" in capsys.readouterr().out + assert not output.exists() + + def test_replay_defaults_partial_summary_profile_failure_mode_to_none(tmp_path): fixture = ReplayFixture( name="partial_summary_profile", diff --git a/tests/test_int8_two_stage_knn.py b/tests/test_int8_two_stage_knn.py index d18c20485..065b90ab3 100644 --- a/tests/test_int8_two_stage_knn.py +++ b/tests/test_int8_two_stage_knn.py @@ -157,6 +157,61 @@ def test_two_stage_reports_full_coverage(tmp_path): vs.close() +def test_in_memory_two_stage_query_with_deadline_uses_current_connection(): + vs = VectorStore(":memory:", bounded_scan_rows=1) + try: + vs.connection.execute( + """ + CREATE TABLE messages ( + store_id INTEGER PRIMARY KEY, + session_id TEXT NOT NULL, + source TEXT DEFAULT '', + role TEXT NOT NULL, + content TEXT, + timestamp REAL NOT NULL + ) + """ + ) + vs.connection.executemany( + "INSERT INTO messages(" + "store_id, session_id, source, role, content, timestamp" + ") VALUES (?, ?, ?, ?, ?, ?)", + [ + (0, "s", "hist", "user", "m", 0.0), + (1, "s", "hist", "user", "m", 1.0), + ], + ) + i8 = _int8_identity(4) + vs.register_profile(MODEL, PROVIDER, 4, dtype="int8", task="chunk") + for idx, vec in enumerate( + ([1.0, 0.0, 0.0, 0.0], [0.0, 1.0, 0.0, 0.0]) + ): + vs.record_chunk_embedding( + f"{idx}:0", + MODEL, + vec, + store_id=idx, + chunk_index=0, + char_start=0, + char_end=1, + token_estimate=1, + identity=i8, + ) + + result = vs.knn_chunks( + [1.0, 0.0, 0.0, 0.0], + k=1, + model=MODEL, + provider=PROVIDER, + deadline=vector_store_module._monotonic() + 5.0, + ) + + assert result.coverage == "full_approx" + assert result[0][0] == "0:0" + finally: + vs.close() + + def test_chunk_deadline_bounds_a_synced_binary_prescreen(tmp_path, monkeypatch): db_path = tmp_path / "lcm.db" _seed_messages(db_path, 2) diff --git a/tests/test_lcm_recall.py b/tests/test_lcm_recall.py index e9cf11d27..9079ec471 100644 --- a/tests/test_lcm_recall.py +++ b/tests/test_lcm_recall.py @@ -581,6 +581,54 @@ def test_answer_ready_delta_refs_match_hits_after_response_cap_eviction( assert not (all_refs - set(delivered_refs)) & set(payload["delta"]["novel_refs"]) +def test_answer_ready_response_cap_evicts_summary_leads_without_delta_refs( + recall_engine, monkeypatch +): + response_cap = 6_000 + summary_leads = [ + { + "node_id": index, + "session_id": f"session-{index}-" + ("s" * 2_000), + "expand_hint": "x" * 2_000, + } + for index in range(3) + ] + monkeypatch.setattr(lcm_tools, "_LCM_RECALL_RESPONSE_CHAR_CAP", response_cap) + monkeypatch.setattr( + lcm_tools, + "_lcm_recall_summary_arm", + lambda *_a, **_k: ( + [], + "full", + len(summary_leads), + len(summary_leads), + list(summary_leads), + ), + ) + monkeypatch.setattr( + lcm_tools, + "_lcm_recall_chunk_arm", + lambda *_a, **_k: ([], "none", 0, 0), + ) + + payload = _recall( + recall_engine, + monkeypatch, + include="summaries", + detail="answer_ready", + limit=3, + seen_refs=[], + ) + + expansion = payload["provenance"]["answer_ready"] + assert payload["hits"] == [] + assert len(expansion["summary_leads"]) < len(summary_leads) + assert len(json.dumps(payload, ensure_ascii=False)) <= response_cap + assert expansion["response_truncated"] is True + assert payload["delta"]["novel_refs"] == [] + assert payload["delta"]["novel_ref_count"] == 0 + + def test_answer_ready_baseline_bytes_ignore_disabled_occurrence_extension( recall_engine, monkeypatch ): diff --git a/tests/test_schema_stamp_remediation.py b/tests/test_schema_stamp_remediation.py index fa89b5fda..c99398b12 100644 --- a/tests/test_schema_stamp_remediation.py +++ b/tests/test_schema_stamp_remediation.py @@ -256,6 +256,52 @@ def test_classify_interim_stamp_with_current_query_and_trajectory_tables(tmp_pat assert _stored_version(db_path) == db_bootstrap.SCHEMA_VERSION +@pytest.mark.parametrize( + "ddl", + ( + """ + CREATE TRIGGER lcm_query_future_trigger + AFTER INSERT ON lcm_query_views + BEGIN + SELECT 1; + END; + """, + """ + CREATE INDEX lcm_query_future_index + ON lcm_query_views(view_id); + """, + ), +) +def test_classify_genuinely_newer_on_unknown_query_schema_object( + tmp_path, ddl +): + db_path = tmp_path / "lcm.db" + _build_v5_db(db_path) + asset_root = tmp_path / "assets" + asset_root.mkdir() + _add_current_query_and_trajectory_tables(db_path, asset_root) + conn = sqlite3.connect(db_path) + try: + conn.executescript(ddl) + conn.commit() + finally: + conn.close() + _stamp(db_path, db_bootstrap.SCHEMA_VERSION + 1) + + conn = sqlite3.connect(db_path) + try: + assert ( + classify_version_mismatch(conn) + == db_bootstrap.VERSION_MISMATCH_GENUINELY_NEWER + ) + result = remediate_interim_schema_stamp(conn, apply=True) + finally: + conn.close() + + assert result["status"] == "refused" + assert _stored_version(db_path) == db_bootstrap.SCHEMA_VERSION + 1 + + @pytest.mark.parametrize( ("table", "column"), ( diff --git a/tests/test_trajectory_compact_delivery.py b/tests/test_trajectory_compact_delivery.py index 9b5eb40fb..f60e6d61a 100644 --- a/tests/test_trajectory_compact_delivery.py +++ b/tests/test_trajectory_compact_delivery.py @@ -254,6 +254,17 @@ def test_c3_typed_queries_exact_title_template_and_budget(tmp_path: Path): assert sharp["rendered_text_tokens_after"] <= budget +def test_question_template_matches_temporal_terms_as_whole_tokens(): + assert ( + TrajectoryStore._question_template("How do I update account settings?") + == "generic" + ) + assert ( + TrajectoryStore._question_template("What changed after yesterday?") + == "temporal" + ) + + def _cap_row( state_id: int, trajectory_id: str, diff --git a/tests/test_trajectory_state_semantic_expansion.py b/tests/test_trajectory_state_semantic_expansion.py index 562ea75ac..76b117b2f 100644 --- a/tests/test_trajectory_state_semantic_expansion.py +++ b/tests/test_trajectory_state_semantic_expansion.py @@ -460,7 +460,9 @@ def test_backfill_is_idempotent_and_resumable(tmp_path): assert provider.document_calls == 0 -def test_profile_activates_only_after_resumed_backfill_completes(tmp_path): +def test_profile_replacement_keeps_old_active_until_resumed_backfill_completes( + tmp_path, +): asset_root = tmp_path / "assets" asset_root.mkdir() initial = StateVectorProvider() @@ -481,6 +483,8 @@ def test_profile_activates_only_after_resumed_backfill_completes(tmp_path): )) store.finalize(["answerpath"]) store.build_state_semantic_index(initial) + prior = store.active_state_semantic_profile() + assert prior is not None interrupted = InterruptingStateVectorProvider( model_id="fake-state-v2", fail_after=1 @@ -490,7 +494,13 @@ def test_profile_activates_only_after_resumed_backfill_completes(tmp_path): interrupted, batch_max_items=1, batch_token_budget=100_000 ) - assert store.active_state_semantic_profile() is None + active_during_backfill = store.active_state_semantic_profile() + assert active_during_backfill is not None + assert active_during_backfill["profile_digest"] == prior["profile_digest"] + assert store._conn.execute( + "SELECT COUNT(*) FROM lcm_trajectory_state_embedding_profiles " + "WHERE active = 1" + ).fetchone()[0] == 1 staged = store._conn.execute( """ SELECT profile_digest, active @@ -500,6 +510,7 @@ def test_profile_activates_only_after_resumed_backfill_completes(tmp_path): ("fake-state-v2",), ).fetchone() assert staged is not None and int(staged["active"]) == 0 + assert staged["profile_digest"] != active_during_backfill["profile_digest"] staged_count = store._conn.execute( "SELECT COUNT(*) FROM lcm_trajectory_state_embeddings " "WHERE profile_digest = ?", @@ -516,6 +527,11 @@ def test_profile_activates_only_after_resumed_backfill_completes(tmp_path): assert resumed["states_embedded"] == 2 assert active is not None assert active["model_name"] == "fake-state-v2" + assert active["profile_digest"] == staged["profile_digest"] + assert store._conn.execute( + "SELECT COUNT(*) FROM lcm_trajectory_state_embedding_profiles " + "WHERE active = 1" + ).fetchone()[0] == 1 def test_dimension_probe_is_bounded_by_document_token_budget(tmp_path): @@ -673,6 +689,49 @@ def record_progress(stats): assert ledger[-1]["states_embedded"] == 0 +def test_dimension_probe_progress_can_stop_before_document_request(tmp_path): + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = StateVectorProvider() + store = TrajectoryStore( + tmp_path / "lcm.db", + _identity(), + asset_root=asset_root, + embedding_provider=provider, + ) + store.insert( + _source( + asset_root, + trajectory_id="probe-cap", + ordinal=0, + goal="Trip the cap after the dimension probe", + texts=("alpha-answer account settings",), + ) + ) + store.finalize(["probe-cap"]) + ledger: list[dict] = [] + + class CostCapExceeded(RuntimeError): + pass + + def record_progress(stats): + ledger.append(dict(stats)) + raise CostCapExceeded("probe spent the test cap") + + with pytest.raises(CostCapExceeded, match="probe spent the test cap"): + store.build_state_semantic_index( + provider, + progress_callback=record_progress, + ) + + assert provider.query_calls == 1 + assert provider.document_calls == 0 + assert len(ledger) == 1 + assert ledger[0]["provider_calls"] == 1 + assert ledger[0]["billed_tokens"] == 1 + assert ledger[0]["states_embedded"] == 0 + + def test_state_query_provider_failure_degrades_to_lexical(tmp_path): provider = StateVectorProvider() store = _build_invisible_semantic_store(tmp_path, provider=provider) diff --git a/tests/test_vector_store.py b/tests/test_vector_store.py index d38c35d9a..57cde0395 100644 --- a/tests/test_vector_store.py +++ b/tests/test_vector_store.py @@ -587,13 +587,22 @@ def test_full_scan_budget_stops_early_and_reports_bounded(tmp_path, monkeypatch) gold = _seed_scan_corpus( dag, store, 6, gold_vector=[1.0, 0.0, 0.0], filler_vector=[0.0, 1.0, 0.0] ) - # A clock that advances a second per reading: the budget is spent after - # the first batch, so the scan stops with 4 of 6 vectors unscored. + # Spend the budget while loading the first batch, so the scan stops with + # 4 of 6 vectors unscored. Candidate enumeration itself stays at t=0; + # its separate regression below proves that it shares this budget. with monkeypatch.context() as clock_patch: - ticks = iter(range(1_000)) + now = [0.0] + original_load = VectorStore._load_vectors_for_ids + + def timed_load(self, *args, **kwargs): + loaded = original_load(self, *args, **kwargs) + now[0] = 1.0 + return loaded + clock_patch.setattr( - vector_store_module, "_monotonic", lambda: float(next(ticks)) + vector_store_module, "_monotonic", lambda: now[0] ) + clock_patch.setattr(VectorStore, "_load_vectors_for_ids", timed_load) result = store.knn( [1.0, 0.0, 0.0], k=1, @@ -611,6 +620,47 @@ def test_full_scan_budget_stops_early_and_reports_bounded(tmp_path, monkeypatch) dag.close() +def test_full_scan_budget_includes_candidate_enumeration(tmp_path, monkeypatch): + db_path = tmp_path / "full-scan-enumeration-budget.db" + dag = SummaryDAG(db_path) + store = VectorStore(db_path, bounded_scan_rows=2) + try: + _seed_scan_corpus( + dag, + store, + 6, + gold_vector=[1.0, 0.0, 0.0], + filler_vector=[0.0, 1.0, 0.0], + ) + now = [0.0] + original = VectorStore._bounded_candidate_ids + + def slow_enumeration(self, *args, **kwargs): + candidate_ids = original(self, *args, **kwargs) + now[0] = 1.0 + return candidate_ids + + monkeypatch.setattr(vector_store_module, "_monotonic", lambda: now[0]) + monkeypatch.setattr( + VectorStore, "_bounded_candidate_ids", slow_enumeration + ) + + result = store.knn( + [1.0, 0.0, 0.0], + k=1, + model="scan", + full_scan=True, + scan_budget_s=0.5, + ) + + assert result.coverage == "bounded" + assert result.scanned == 0 + assert result == [] + finally: + store.close() + dag.close() + + def test_full_scan_absolute_deadline_stops_between_batches(tmp_path, monkeypatch): """The operation deadline remains a hard stop when the relative scan budget is disabled (zero), so recall cannot start another full batch after expiry.""" @@ -622,10 +672,18 @@ def test_full_scan_absolute_deadline_stops_between_batches(tmp_path, monkeypatch dag, store, 6, gold_vector=[1.0, 0.0, 0.0], filler_vector=[0.0, 1.0, 0.0] ) with monkeypatch.context() as clock_patch: - ticks = iter((0.0, 0.0, 2.0)) + now = [0.0] + original_load = VectorStore._load_vectors_for_ids + + def timed_load(self, *args, **kwargs): + loaded = original_load(self, *args, **kwargs) + now[0] = 2.0 + return loaded + clock_patch.setattr( - vector_store_module, "_monotonic", lambda: next(ticks) + vector_store_module, "_monotonic", lambda: now[0] ) + clock_patch.setattr(VectorStore, "_load_vectors_for_ids", timed_load) result = store.knn( [1.0, 0.0, 0.0], k=1, diff --git a/tools.py b/tools.py index cf3ed2aa3..e2262bd53 100644 --- a/tools.py +++ b/tools.py @@ -5186,8 +5186,13 @@ def _novel(entries: list[dict[str, Any]]) -> list[dict[str, Any]]: response["query"] = original_query[:4_096] expansion["query_truncated"] = len(response["query"]) < len(original_query) encoded = json.dumps(response, ensure_ascii=False) - while len(encoded) > _LCM_RECALL_RESPONSE_CHAR_CAP and response["hits"]: - response["hits"].pop() + while len(encoded) > _LCM_RECALL_RESPONSE_CHAR_CAP and ( + response["hits"] or expansion.get("summary_leads") + ): + if response["hits"]: + response["hits"].pop() + else: + expansion["summary_leads"].pop() response["total_results"] = len(response["hits"]) expansion["response_truncated"] = True expansion["expanded_hit_count"] = sum( diff --git a/trajectory_store.py b/trajectory_store.py index 754381e1e..2e90f9b2e 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -1692,10 +1692,6 @@ def build_state_semantic_index( with self._lock: self._conn.execute("BEGIN IMMEDIATE") try: - self._conn.execute( - "UPDATE lcm_trajectory_state_embedding_profiles " - "SET active = 0 WHERE active = 1", - ) self._conn.execute( """ INSERT INTO lcm_trajectory_state_embedding_profiles( @@ -1746,6 +1742,13 @@ def build_state_semantic_index( "billed_tokens": probe_billed_tokens, } + def _emit_progress() -> None: + if progress_callback is not None: + progress_callback(dict(stats)) + + if probe_provider_calls: + _emit_progress() + # Partition pending states into single-request documents and the # oversize (chunked) minority, both packed to the same item/token caps. normal: list[tuple[int, str, str, int]] = [] # (state_id, doc, sha, tokens) @@ -1785,10 +1788,6 @@ def _persist(rows: list[tuple[int, str, bytes]]) -> None: self._conn.rollback() raise - def _emit_progress() -> None: - if progress_callback is not None: - progress_callback(dict(stats)) - # --- normal single-request states ------------------------------------- batch_ids: list[int] = [] batch_docs: list[str] = [] @@ -2717,7 +2716,8 @@ def _question_template(query: str) -> str: folded = query.casefold() if any(term in folded for term in ("protocol", "workflow", "procedure", "steps")): return "procedure" - if any(term in folded for term in _TEMPORAL_TERMS): + words = set(re.findall(r"[a-z0-9]+", folded)) + if words.intersection(_TEMPORAL_TERMS): return "temporal" if ( any(character in query for character in ",;") diff --git a/vector_store.py b/vector_store.py index 89a710c01..3038116b9 100644 --- a/vector_store.py +++ b/vector_store.py @@ -1924,7 +1924,6 @@ def _load_binary_with_deadline(self, loader, deadline: float | None): remaining = deadline - _monotonic() if remaining <= 0: raise _PrescreenDeadlineExpired() - uri = f"{self.db_path.resolve().as_uri()}?mode=ro" conn: sqlite3.Connection | None = None expired = [False] @@ -1934,6 +1933,29 @@ def interrupt_if_expired() -> int: return 1 return 0 + db_path = str(self.db_path) + query = db_path.partition("?")[2] + in_memory = ( + db_path == ":memory:" + or db_path.startswith("file::memory:") + or any(part == "mode=memory" for part in query.split("&")) + ) + if in_memory: + with self._write_lock: + try: + self._conn.set_progress_handler(interrupt_if_expired, 1000) + loaded = loader(self) + if _monotonic() >= deadline: + raise _PrescreenDeadlineExpired() + return loaded + except sqlite3.OperationalError as exc: + if expired[0] or _monotonic() >= deadline: + raise _PrescreenDeadlineExpired() from exc + raise + finally: + self._conn.set_progress_handler(None, 0) + + uri = f"{self.db_path.resolve().as_uri()}?mode=ro" try: conn = sqlite3.connect( uri, @@ -1959,6 +1981,10 @@ def interrupt_if_expired() -> int: if conn is not None: conn.close() + def _read_with_deadline(self, loader, deadline: float | None): + """Run a candidate read through the shared interruptible read path.""" + return self._load_binary_with_deadline(loader, deadline) + @staticmethod def _binary_rows_to_matrix(numpy: Any, rows: Sequence[sqlite3.Row]) -> tuple[list[str], Any]: ids: list[str] = [] @@ -2067,6 +2093,7 @@ def knn( scan_budget_s: float = 0.0, deadline: float | None = None, ) -> KNNResult: + operation_started = _monotonic() k = int(k) if k <= 0: return KNNResult(coverage="none") @@ -2164,20 +2191,37 @@ def knn( probe_limit, scan_limit = self._scan_limits( full_scan=full_scan, scan_max_rows=scan_max_rows ) + scan_deadline = deadline + if scan_budget_s > 0: + budget_deadline = operation_started + scan_budget_s + scan_deadline = ( + budget_deadline + if scan_deadline is None + else min(scan_deadline, budget_deadline) + ) # Probe one past the scan limit through the indexed candidate query. # This determines full-vs-bounded coverage without COUNT(*) scanning the # entire identity on every request. try: - probed_ids = self._bounded_candidate_ids( - identity, - since=since, - until=until, - conversation_ids=conversation_ids, - source=source, - limit=probe_limit, + probed_ids = self._read_with_deadline( + lambda reader: reader._bounded_candidate_ids( + identity, + since=since, + until=until, + conversation_ids=conversation_ids, + source=source, + limit=probe_limit, + ), + scan_deadline, ) except _UnverifiableProvenance: return KNNResult(coverage="none", reason="unverifiable_provenance") + except _PrescreenDeadlineExpired as exc: + return KNNResult( + coverage="bounded", + scanned=exc.scanned, + total=self._count_embedded_vectors(identity, chunk=False), + ) if not probed_ids: return KNNResult(coverage="none") scan_ids = probed_ids if scan_limit is None else probed_ids[:scan_limit] @@ -2227,7 +2271,7 @@ def score_batch( candidate_ids=scan_ids, batch_rows=max(1, self.bounded_scan_rows), budget_s=scan_budget_s, - deadline=deadline, + deadline=scan_deadline, limit=k, score_batch=score_batch, ) @@ -2702,6 +2746,7 @@ def knn_chunks( cut short by a hard cap, latency budget, or operation deadline, and ``full`` when the requested candidate set was scanned completely. """ + operation_started = _monotonic() k = int(k) if k <= 0: return KNNResult(coverage="none") @@ -2798,14 +2843,32 @@ def knn_chunks( probe_limit, scan_limit = self._scan_limits( full_scan=full_scan, scan_max_rows=scan_max_rows ) - probed_ids = self._bounded_chunk_candidate_ids( - identity, - since=since, - until=until, - conversation_ids=conversation_ids, - source=source, - limit=probe_limit, - ) + scan_deadline = deadline + if scan_budget_s > 0: + budget_deadline = operation_started + scan_budget_s + scan_deadline = ( + budget_deadline + if scan_deadline is None + else min(scan_deadline, budget_deadline) + ) + try: + probed_ids = self._read_with_deadline( + lambda reader: reader._bounded_chunk_candidate_ids( + identity, + since=since, + until=until, + conversation_ids=conversation_ids, + source=source, + limit=probe_limit, + ), + scan_deadline, + ) + except _PrescreenDeadlineExpired as exc: + return KNNResult( + coverage="bounded", + scanned=exc.scanned, + total=self._count_embedded_vectors(identity, chunk=True), + ) if not probed_ids: return KNNResult(coverage="none") scan_ids = probed_ids if scan_limit is None else probed_ids[:scan_limit] @@ -2841,7 +2904,7 @@ def score_batch( candidate_ids=scan_ids, batch_rows=max(1, self.bounded_scan_rows), budget_s=scan_budget_s, - deadline=deadline, + deadline=scan_deadline, limit=k, score_batch=score_batch, ) From 4b684c9f39c1e4e53866c70d26fe70aeebb02a81 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 14:11:09 +0700 Subject: [PATCH 52/54] round-5 review fixes: revert unsound continuity change; close the index-gate gap; real test seam (PR #175) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - REVERT round-4's keep-old-profile-active-during-staging: the reviewer's deeper pass showed the vector upsert conflicts on state_id alone, so a staged rebuild progressively overwrites the previous profile's vectors — an active predecessor would serve a MIXED index. Round-4's premise (the deactivation 'serves no purpose') was wrong given that schema; staging deactivates up front again and readers degrade to lexical until atomic cutover. Rebuild availability needs profile-scoped embeddings ((profile_digest, state_id) identity) — filed as the next-train issue. - query_view verifier: index gating now uses the real idx_lcm_query naming prefix — the index half of the fail-close gate could never match a legitimately-named rogue index (probe fixture corrected to match). - binary-prescreen deadline checks go through a position-aware seam (_prescreen_deadline_expired(deadline, scanned_rows)); the two deadline tests patch that seam keyed on rows-scanned instead of frame inspection or process-wide clock stubs, with genuinely-future deadlines so only the seam trips. Author + review: orchestrator (round-5 inline batch; bots confirm next pass). Touched area 142 green; full suite exact-name parity (35 known). --- query_view_store.py | 6 ++++- tests/test_int8_two_stage_knn.py | 20 +++++++--------- tests/test_prescreen_flip_blackout.py | 24 +++++++++---------- tests/test_schema_stamp_remediation.py | 2 +- ...est_trajectory_state_semantic_expansion.py | 16 ++++++++----- trajectory_store.py | 11 +++++++++ vector_store.py | 18 ++++++++++++-- 7 files changed, 63 insertions(+), 34 deletions(-) diff --git a/query_view_store.py b/query_view_store.py index 69ece2639..4d3a25301 100644 --- a/query_view_store.py +++ b/query_view_store.py @@ -517,7 +517,11 @@ def _verify_query_view_schema(conn: sqlite3.Connection) -> list[str]: missing.extend( f"{object_type}:{name}" for name in sorted(names - actual) ) - gated = {name for name in actual if name.startswith("lcm_query")} + # Real index names carry the idx_ prefix (idx_lcm_query_*); gating on + # the bare table prefix would make the index half of this fail-closed + # check unable to match any legitimately-named extra index. + gate_prefix = "idx_lcm_query" if object_type == "index" else "lcm_query" + gated = {name for name in actual if name.startswith(gate_prefix)} missing.extend( f"unexpected-{object_type}:{name}" for name in sorted(gated - names) diff --git a/tests/test_int8_two_stage_knn.py b/tests/test_int8_two_stage_knn.py index 065b90ab3..20059a8cb 100644 --- a/tests/test_int8_two_stage_knn.py +++ b/tests/test_int8_two_stage_knn.py @@ -9,7 +9,6 @@ from __future__ import annotations import sqlite3 -import sys import numpy as np import hermes_lcm.vector_store as vector_store_module @@ -27,15 +26,10 @@ PROVIDER = "voyage" -def _expire_after_first_prescreen_batch() -> float: - frame = sys._getframe(1) - return ( - 2.0 - if frame.f_code.co_name == "_two_stage_rank" - and frame.f_locals.get("start") == 0 - and frame.f_locals.get("end") == 1 - else 0.0 - ) +def _expire_after_first_prescreen_batch(deadline, scanned_rows): + """Position-keyed deadline stub: expire once the first prescreen batch + has completed (scanned_rows >= 1), regardless of clock call patterns.""" + return deadline is not None and scanned_rows >= 1 def _seed_messages(db_path, count, *, session="s", source="hist", ts0=0.0): @@ -239,7 +233,7 @@ def test_chunk_deadline_bounds_a_synced_binary_prescreen(tmp_path, monkeypatch): ) monkeypatch.setattr( vector_store_module, - "_monotonic", + "_prescreen_deadline_expired", _expire_after_first_prescreen_batch, ) expired = vs.knn_chunks( @@ -248,7 +242,9 @@ def test_chunk_deadline_bounds_a_synced_binary_prescreen(tmp_path, monkeypatch): model=MODEL, provider=PROVIDER, full_scan=True, - deadline=1.0, + # Genuinely-future deadline: only the patched position-keyed seam + # trips; the real clock never expires the load path. + deadline=vector_store_module._monotonic() + 60.0, ) assert unbounded.coverage == "full_approx" diff --git a/tests/test_prescreen_flip_blackout.py b/tests/test_prescreen_flip_blackout.py index 1d830d967..f43c13cff 100644 --- a/tests/test_prescreen_flip_blackout.py +++ b/tests/test_prescreen_flip_blackout.py @@ -16,7 +16,6 @@ """ from __future__ import annotations -import sys import hermes_lcm.vector_store as vector_store_module from hermes_lcm.config import LCMConfig @@ -29,15 +28,10 @@ DIM = 3 -def _expire_after_first_prescreen_batch() -> float: - frame = sys._getframe(1) - return ( - 2.0 - if frame.f_code.co_name == "_two_stage_rank" - and frame.f_locals.get("start") == 0 - and frame.f_locals.get("end") == 1 - else 0.0 - ) +def _expire_after_first_prescreen_batch(deadline, scanned_rows): + """Position-keyed deadline stub: expire once the first prescreen batch + has completed (scanned_rows >= 1), regardless of clock call patterns.""" + return deadline is not None and scanned_rows >= 1 def _add_summary(dag: SummaryDAG, *, created_at: float) -> int: @@ -205,11 +199,17 @@ def test_deadline_bounds_a_synced_binary_summary_prescreen(tmp_path, monkeypatch unbounded = store.knn([1.0, 0.0, 0.0], k=1, model=MODEL, full_scan=True) monkeypatch.setattr( vector_store_module, - "_monotonic", + "_prescreen_deadline_expired", _expire_after_first_prescreen_batch, ) expired = store.knn( - [1.0, 0.0, 0.0], k=1, model=MODEL, full_scan=True, deadline=1.0 + [1.0, 0.0, 0.0], + k=1, + model=MODEL, + full_scan=True, + # A genuinely-future deadline: only the patched position-keyed seam + # trips; the real clock never expires the load path. + deadline=vector_store_module._monotonic() + 60.0, ) assert unbounded.coverage == "full_approx" diff --git a/tests/test_schema_stamp_remediation.py b/tests/test_schema_stamp_remediation.py index c99398b12..38c961b44 100644 --- a/tests/test_schema_stamp_remediation.py +++ b/tests/test_schema_stamp_remediation.py @@ -267,7 +267,7 @@ def test_classify_interim_stamp_with_current_query_and_trajectory_tables(tmp_pat END; """, """ - CREATE INDEX lcm_query_future_index + CREATE INDEX idx_lcm_query_future ON lcm_query_views(view_id); """, ), diff --git a/tests/test_trajectory_state_semantic_expansion.py b/tests/test_trajectory_state_semantic_expansion.py index 76b117b2f..4204bb573 100644 --- a/tests/test_trajectory_state_semantic_expansion.py +++ b/tests/test_trajectory_state_semantic_expansion.py @@ -460,9 +460,15 @@ def test_backfill_is_idempotent_and_resumable(tmp_path): assert provider.document_calls == 0 -def test_profile_replacement_keeps_old_active_until_resumed_backfill_completes( +def test_interrupted_profile_rebuild_leaves_no_active_profile_until_cutover( tmp_path, ): + # The vector upsert conflicts on state_id alone, so a staged rebuild + # overwrites the previous profile's vectors as it goes. The previous + # profile therefore CANNOT keep serving during a rebuild (it would serve + # a mixed index) — staging deactivates it up front, and readers degrade + # to lexical until the atomic cutover at completion. Profile-scoped + # embeddings are the tracked prerequisite for rebuild availability. asset_root = tmp_path / "assets" asset_root.mkdir() initial = StateVectorProvider() @@ -494,13 +500,11 @@ def test_profile_replacement_keeps_old_active_until_resumed_backfill_completes( interrupted, batch_max_items=1, batch_token_budget=100_000 ) - active_during_backfill = store.active_state_semantic_profile() - assert active_during_backfill is not None - assert active_during_backfill["profile_digest"] == prior["profile_digest"] + assert store.active_state_semantic_profile() is None assert store._conn.execute( "SELECT COUNT(*) FROM lcm_trajectory_state_embedding_profiles " "WHERE active = 1" - ).fetchone()[0] == 1 + ).fetchone()[0] == 0 staged = store._conn.execute( """ SELECT profile_digest, active @@ -510,7 +514,7 @@ def test_profile_replacement_keeps_old_active_until_resumed_backfill_completes( ("fake-state-v2",), ).fetchone() assert staged is not None and int(staged["active"]) == 0 - assert staged["profile_digest"] != active_during_backfill["profile_digest"] + assert staged["profile_digest"] != prior["profile_digest"] staged_count = store._conn.execute( "SELECT COUNT(*) FROM lcm_trajectory_state_embeddings " "WHERE profile_digest = ?", diff --git a/trajectory_store.py b/trajectory_store.py index 2e90f9b2e..d3756673c 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -1692,6 +1692,17 @@ def build_state_semantic_index( with self._lock: self._conn.execute("BEGIN IMMEDIATE") try: + # Deactivate the serving profile BEFORE staging: the vector + # upsert below conflicts on state_id alone, so a staged rebuild + # progressively overwrites the previous profile's vectors — an + # "active" predecessor would keep serving a mixed index. + # Availability-during-rebuild needs profile-scoped embeddings + # ((profile_digest, state_id) identity) first; tracked in the + # fork issue for the next train. + self._conn.execute( + "UPDATE lcm_trajectory_state_embedding_profiles " + "SET active = 0 WHERE active = 1", + ) self._conn.execute( """ INSERT INTO lcm_trajectory_state_embedding_profiles( diff --git a/vector_store.py b/vector_store.py index 3038116b9..8ca5ba0a5 100644 --- a/vector_store.py +++ b/vector_store.py @@ -41,6 +41,20 @@ _monotonic = time.monotonic + +def _prescreen_deadline_expired(deadline: float | None, scanned_rows: int) -> bool: + """Deadline check for the binary-prescreen loop, position-aware. + + ``scanned_rows`` is how many rows have completed when the check runs (the + batch's ``start`` before scoring, its ``end`` after). Tests patch THIS + seam keyed on position — not the process clock, call counts, or frame + inspection — so a loop refactor cannot silently disable the covered + branch. + """ + del scanned_rows + return deadline is not None and _monotonic() >= deadline + + # Vectors are float32 in native little-endian order by default. These are # recorded as part of the canonical profile identity so a dtype/byteorder change # is detectable rather than silently reinterpreting stored bytes. @@ -2043,14 +2057,14 @@ def _two_stage_rank( hamming = numpy.empty(n, dtype=numpy.uint32) prescreen_rows = min(max(1, self.bounded_scan_rows), 4096) for start in range(0, n, prescreen_rows): - if deadline is not None and _monotonic() >= deadline: + if _prescreen_deadline_expired(deadline, start): raise _PrescreenDeadlineExpired(start) end = min(n, start + prescreen_rows) xor = numpy.bitwise_xor( binary_matrix[start:end, :width], query_bits[:width] ) hamming[start:end] = popcount[xor].sum(axis=1) - if deadline is not None and _monotonic() >= deadline: + if _prescreen_deadline_expired(deadline, end): raise _PrescreenDeadlineExpired(end) m = min(n, max(k, self.knn_prescreen_multiplier * k)) if m < n: From 05a46f62ea5deb913efa1b3a9db13e8829426f43 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 14:52:07 +0700 Subject: [PATCH 53/54] round-6 review fixes: confirmation-pass findings (PR #175) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - adaptive_retrieval: requirements-digest capacity is DERIVED from the declared limits (MAX_REQUIREMENTS x worst-case canonical item, NUL-escape modeled, one item headroom) — a fully-validated 12x max-sized request no longer fails at start() after passing field validation (P1, found independently by both reviewers; introduced when description joined the digest identity) - trajectory: normal-batch provider spend ledgers BEFORE failure-prone validation/persistence (the probe/chunk paths' guarantee, now on the third call site) - trajectory: termless queries (emoji-only) become an empty lexical pool instead of early-returning past the state-semantic arm when it's enabled (second call site of the round-1 lexical-miss principle) - h5 replay: output dirs created before any paid provider work Author: codex sol-high (R6); review: orchestrator. Full-suite exact-name parity; 4 focused regressions green. --- FINDINGS-VERDICTS-R6.md | 8 +++ adaptive_retrieval.py | 48 ++++++++++++-- benchmarking/h5_state_semantic_replay.py | 1 + tests/test_adaptive_retrieval.py | 29 +++++++++ tests/test_h5_state_semantic_replay.py | 19 ++++++ ...est_trajectory_state_semantic_expansion.py | 65 +++++++++++++++++++ trajectory_store.py | 17 +++-- 7 files changed, 174 insertions(+), 13 deletions(-) create mode 100644 FINDINGS-VERDICTS-R6.md create mode 100644 tests/test_h5_state_semantic_replay.py diff --git a/FINDINGS-VERDICTS-R6.md b/FINDINGS-VERDICTS-R6.md new file mode 100644 index 000000000..07fff5ec5 --- /dev/null +++ b/FINDINGS-VERDICTS-R6.md @@ -0,0 +1,8 @@ +# Findings verdicts — R6 + +- FIXED — adaptive_retrieval.py P1 (Codex 3671928491; EvaOS 3671869848): digest capacity is derived from all declared requirement limits with one max-item headroom; 12 max-sized requirements validate and start. +- FIXED — trajectory_store.py P2 (Codex 3671928473): normal-batch provider spend and progress emit before vector validation/persistence; forced persist failure retains the full spend ledger. +- FIXED — trajectory_store.py P2 (Codex 3671928480): termless lexical queries become an empty lexical pool when state semantics are enabled; emoji-only semantic recall succeeds and quota-zero telemetry stays identical. +- FIXED — benchmarking/h5_state_semantic_replay.py P2 (Codex 3671928484): output parents are created immediately after argument parsing, before provider construction, the golden gate, warm-up, or sweep. +- VALIDATION — 4 focused regressions and 38 touched-area tests pass; Ruff passes; full suite has 35/35 baseline failure names with zero new names. +- PROOF BOUNDARY — source and CI-replica local test proof only; no commit, push, PR update, merge, release, or runtime change. diff --git a/adaptive_retrieval.py b/adaptive_retrieval.py index 15a7e9d23..e3fd7cee6 100644 --- a/adaptive_retrieval.py +++ b/adaptive_retrieval.py @@ -36,6 +36,8 @@ MAX_CONTEXT_TOKENS = 7_500 MAX_CONTEXT_CHARS = 40_000 MAX_REQUIREMENTS = 12 +MAX_REQUIREMENT_SLOT_ID_CHARS = 64 +MAX_REQUIREMENT_DESCRIPTION_CHARS = 256 MAX_ACTIVE_RETRIEVALS = 32 RETRIEVAL_TTL_SECONDS = 15 * 60 MAX_EVIDENCE_CHARS = 2_400 @@ -43,7 +45,9 @@ MAX_TOOL_ARGS_CHARS = 2_048 MAX_LEAD_FIELD_CHARS = 512 -_SLOT_ID_RE = re.compile(r"^[a-z][a-z0-9_.-]{0,63}$") +_SLOT_ID_RE = re.compile( + rf"^[a-z][a-z0-9_.-]{{0,{MAX_REQUIREMENT_SLOT_ID_CHARS - 1}}}$" +) _SHA256_RE = re.compile(r"^[0-9a-f]{64}$") _FORBIDDEN_ARGUMENT_KEYS = frozenset({ "question_id", @@ -100,6 +104,30 @@ def _canonical_json(value: Any, *, field_name: str, max_chars: int) -> str: return encoded +# Derived from MAX_REQUIREMENTS, the declared slot/description limits, and +# MAX_CANDIDATE_REFS. NUL models the widest canonical JSON escape; one additional +# max-sized item provides headroom if the identity shape gains small metadata. +_REQUIREMENT_IDENTITY_ITEM_MAX_CHARS = len( + json.dumps( + { + "slot_id": "s" * MAX_REQUIREMENT_SLOT_ID_CHARS, + "description": "\0" * MAX_REQUIREMENT_DESCRIPTION_CHARS, + "minimum_refs": MAX_CANDIDATE_REFS, + }, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ) +) +_REQUIREMENTS_DIGEST_MAX_CHARS = ( + 2 + + MAX_REQUIREMENTS * _REQUIREMENT_IDENTITY_ITEM_MAX_CHARS + + (MAX_REQUIREMENTS - 1) + + _REQUIREMENT_IDENTITY_ITEM_MAX_CHARS +) + + def _normalize_question_date(value: Any) -> str: text = str(value or "").strip().replace("/", "-") if not text: @@ -145,8 +173,14 @@ def parse(cls, value: Any) -> "EvidenceRequirement": "letters, digits, dots, underscores, or hyphens" ) description = " ".join(str(value.get("description") or "").split()) - if not description or len(description) > 256: - raise ValueError("requirement description must contain 1..256 characters") + if ( + not description + or len(description) > MAX_REQUIREMENT_DESCRIPTION_CHARS + ): + raise ValueError( + "requirement description must contain " + f"1..{MAX_REQUIREMENT_DESCRIPTION_CHARS} characters" + ) raw_minimum = value.get("minimum_refs", 1) if isinstance(raw_minimum, bool): raise ValueError("minimum_refs must be an integer") @@ -180,9 +214,11 @@ def requirements_digest(requirements: Sequence[EvidenceRequirement]) -> str: for item in sorted(requirements, key=lambda item: item.slot_id) ] return hashlib.sha256( - _canonical_json(identity, field_name="requirements", max_chars=4_096).encode( - "utf-8" - ) + _canonical_json( + identity, + field_name="requirements", + max_chars=_REQUIREMENTS_DIGEST_MAX_CHARS, + ).encode("utf-8") ).hexdigest() diff --git a/benchmarking/h5_state_semantic_replay.py b/benchmarking/h5_state_semantic_replay.py index 85b99b38d..b13293646 100644 --- a/benchmarking/h5_state_semantic_replay.py +++ b/benchmarking/h5_state_semantic_replay.py @@ -146,6 +146,7 @@ def main() -> int: parser.add_argument("--latency-sample", type=int, default=50) parser.add_argument("--quotas", type=int, nargs="+", default=[4, 8, 16]) args = parser.parse_args() + args.out.parent.mkdir(parents=True, exist_ok=True) if not os.environ.get("VOYAGE_API_KEY", "").strip(): print("VOYAGE_API_KEY is not set -- the state arm needs a query embedder", diff --git a/tests/test_adaptive_retrieval.py b/tests/test_adaptive_retrieval.py index 07fc0cde1..1ca4c7c72 100644 --- a/tests/test_adaptive_retrieval.py +++ b/tests/test_adaptive_retrieval.py @@ -11,6 +11,7 @@ MAX_CANDIDATE_REFS, MAX_CONTEXT_CHARS, MAX_CONTEXT_TOKENS, + MAX_REQUIREMENTS, MAX_RETRIEVAL_ROUNDS, EvidenceRequirement, requirements_digest, @@ -284,6 +285,34 @@ def test_requirements_digest_distinguishes_descriptions(): assert requirements_digest([ceo]) != requirements_digest([cfo]) +def test_max_sized_requirements_validate_and_digest(tmp_path): + requirements = [ + { + "slot_id": f"slot{index:02d}" + "x" * (64 - len(f"slot{index:02d}")), + "description": "d" * 256, + "minimum_refs": 1, + } + for index in range(MAX_REQUIREMENTS) + ] + + parsed = [EvidenceRequirement.parse(item) for item in requirements] + + assert len({item.slot_id for item in parsed}) == MAX_REQUIREMENTS + assert len(requirements_digest(parsed)) == 64 + engine = _engine(tmp_path) + try: + started = _call( + engine, + action="start", + question="Find the requested evidence.", + identity=_identity(), + requirements=requirements, + ) + assert started["status"] == "active" + finally: + engine.shutdown() + + def test_cached_view_is_not_reused_across_different_requirement_descriptions(tmp_path): """Same identity + same slot_id + same minimum_refs, but a DIFFERENT requirement description, must be a cache miss -- otherwise start() diff --git a/tests/test_h5_state_semantic_replay.py b/tests/test_h5_state_semantic_replay.py new file mode 100644 index 000000000..b46e567c8 --- /dev/null +++ b/tests/test_h5_state_semantic_replay.py @@ -0,0 +1,19 @@ +from __future__ import annotations + +import sys + +from benchmarking import h5_state_semantic_replay + + +def test_output_parent_exists_before_provider_work(tmp_path, monkeypatch): + output = tmp_path / "new" / "nested" / "sweep.json" + monkeypatch.delenv("VOYAGE_API_KEY", raising=False) + monkeypatch.setattr( + sys, + "argv", + ["h5_state_semantic_replay.py", "--out", str(output)], + ) + + assert h5_state_semantic_replay.main() == 3 + assert output.parent.is_dir() + assert not output.exists() diff --git a/tests/test_trajectory_state_semantic_expansion.py b/tests/test_trajectory_state_semantic_expansion.py index 4204bb573..fb0d52f83 100644 --- a/tests/test_trajectory_state_semantic_expansion.py +++ b/tests/test_trajectory_state_semantic_expansion.py @@ -16,6 +16,7 @@ import hashlib from pathlib import Path +import sqlite3 import struct import pytest @@ -259,6 +260,26 @@ def test_expansion_can_fill_a_pure_lexical_miss(tmp_path): ) +def test_expansion_can_fill_a_termless_lexical_miss(tmp_path): + store = _build_invisible_semantic_store(tmp_path) + + baseline = store.query("🚀", image_limit=0) + baseline_telemetry = store.last_query_telemetry() + explicit_off = store.query("🚀", image_limit=0, state_semantic_quota=0) + assert explicit_off == baseline == () + assert store.last_query_telemetry() == baseline_telemetry + + expanded = store.query( + "🚀", + image_limit=0, + include_adjacent=False, + state_semantic_quota=1, + ) + + assert len(expanded) == 1 + assert expanded[0].match_kind == "state_semantic" + + def test_quota_caps_admissions(tmp_path): """Two lexically-invisible alpha states; a quota of 1 admits exactly one.""" asset_root = tmp_path / "assets" @@ -693,6 +714,50 @@ def record_progress(stats): assert ledger[-1]["states_embedded"] == 0 +def test_normal_batch_persist_failure_still_ledgers_spend(tmp_path): + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = StateVectorProvider() + store = TrajectoryStore( + tmp_path / "lcm.db", + _identity(), + asset_root=asset_root, + embedding_provider=provider, + ) + store.insert( + _source( + asset_root, + trajectory_id="persist-failure", + ordinal=0, + goal="Preserve spend before persistence", + texts=("alpha-answer account settings",), + ) + ) + store.finalize(["persist-failure"]) + store._ensure_state_semantic_schema() + store._conn.execute( + """ + CREATE TRIGGER fail_state_embedding_insert + BEFORE INSERT ON lcm_trajectory_state_embeddings + BEGIN + SELECT RAISE(FAIL, 'simulated persist failure'); + END + """ + ) + ledger: list[dict] = [] + + with pytest.raises(sqlite3.IntegrityError, match="simulated persist failure"): + store.build_state_semantic_index( + provider, + progress_callback=lambda stats: ledger.append(dict(stats)), + ) + + assert provider.document_calls == 1 + assert ledger[-1]["provider_calls"] == provider.query_calls + provider.document_calls + assert ledger[-1]["billed_tokens"] == provider.usage_tokens_total + assert ledger[-1]["states_embedded"] == 0 + + def test_dimension_probe_progress_can_stop_before_document_request(tmp_path): asset_root = tmp_path / "assets" asset_root.mkdir() diff --git a/trajectory_store.py b/trajectory_store.py index d3756673c..930b92fc7 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -1810,6 +1810,11 @@ def _flush_normal() -> None: if not batch_docs: return vectors = active_provider.embed_documents(batch_docs) + stats["provider_calls"] += 1 + stats["billed_tokens"] += max( + 0, int(getattr(active_provider, "last_usage_tokens", 0) or 0) + ) + _emit_progress() if len(vectors) != len(batch_docs): raise ValueError("state embedding count does not match batch size") packed = [] @@ -1817,10 +1822,6 @@ def _flush_normal() -> None: normalized = _normalized_vector(vector, expected_dim=dim) packed.append((sid, sha, _pack_vector(normalized))) _persist(packed) - stats["provider_calls"] += 1 - stats["billed_tokens"] += max( - 0, int(getattr(active_provider, "last_usage_tokens", 0) or 0) - ) stats["states_embedded"] += len(packed) batch_ids, batch_docs, batch_shas, batch_tokens = [], [], [], 0 _emit_progress() @@ -3229,7 +3230,7 @@ def query( _MAX_QUERY_TEXT_CHARS, ) expression = self._fts_expression(query) - if not expression: + if not expression and state_semantic_quota == 0: self._last_query_telemetry = { "semantic_attempt": None, "source_candidate_ranks": [], @@ -3238,7 +3239,9 @@ def query( } return () sharp_telemetry: dict[str, Any] | None = None - if sharp_token_budget > 0: + if not expression: + global_rows = [] + elif sharp_token_budget > 0: global_rows, sharp_telemetry = self._sharp_fts_rows( query, candidate_limit ) @@ -3295,7 +3298,7 @@ def query( expression, candidate_limit, source_ids=[source_id for source_id, _score in semantic_ranks], - ) if semantic_ranks else [] + ) if expression and semantic_ranks else [] if semantic_ranks and scoped_rows: row_by_id: dict[int, sqlite3.Row] = {} From 68b1b559791539d347d0355fd843ea2cc76b7886 Mon Sep 17 00:00:00 2001 From: 100yenadmin Date: Wed, 29 Jul 2026 15:29:22 +0700 Subject: [PATCH 54/54] round-7 review fixes: final in-train batch (PR #175) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - trajectory: forced same-profile rebuild (resume=False) clears prior rows before refill — no stale vectors from an earlier fill survive - trajectory: document chunk-routing honors the SMALLER of document/ request token budgets - trajectory: termless queries skip the source-semantic embed (scoped rows are structurally empty there); state-semantic seeding intact - adaptive_retrieval: requirement descriptions reject surrogate code points before canonical UTF-8 digesting - h3 replay: Policy D arm-quota cells measure recoverability from their fused candidate pool (scoped semantic rows included) - h5 test: out-dir contract now proves provider construction is skipped All items quota-gated / bench-side / test-side — no delivery-path change. Author: codex sol-high (R7); review: orchestrator. Full-suite exact-name parity; 6 focused regressions green. STOPPING RULE (declared): this closes the in-train fix cycle. Subsequent non-delivery-path findings are filed as next-train issues. --- FINDINGS-VERDICTS-R7.md | 13 +++ adaptive_retrieval.py | 4 + benchmarking/h3_composition_replay.py | 39 ++++++++- tests/test_adaptive_retrieval.py | 9 ++ tests/test_h3_composition_replay.py | 38 +++++++++ tests/test_h5_state_semantic_replay.py | 9 ++ ...est_trajectory_state_semantic_expansion.py | 83 ++++++++++++++++++ trajectory_store.py | 84 +++++++++++-------- 8 files changed, 239 insertions(+), 40 deletions(-) create mode 100644 FINDINGS-VERDICTS-R7.md create mode 100644 tests/test_h3_composition_replay.py diff --git a/FINDINGS-VERDICTS-R7.md b/FINDINGS-VERDICTS-R7.md new file mode 100644 index 000000000..f05ff276f --- /dev/null +++ b/FINDINGS-VERDICTS-R7.md @@ -0,0 +1,13 @@ +# Round 7 findings verdicts +1. Fixed: forced same-profile rebuilds delete prior profile rows before refill; partial-rebuild regression passes. +2. Fixed: state documents use the smaller document/request token budget for chunk routing; request-cap regression passes. +3. Fixed: Policy D arm-quota cells measure recoverability from their fused candidate pool; scoped-row regression passes. +4. Fixed: evidence requirement descriptions reject surrogate code points before canonical UTF-8 digesting; regression passes. +5. Fixed: the H5 out-dir test now fails if provider construction occurs before the API-key guard. +6. Fixed: termless queries skip source-semantic embedding while state-semantic seeding and telemetry remain intact. +Validation: 6 focused R7 regressions passed. +Validation: full suite 2705 passed, 35 failed, 1 skipped, 12 xfailed. +Baseline comparison: 35/35 prior failure names; 0 new and 0 missing — PASS. +Issue-filing note: generation-scoped/profile-scoped embedding rows remain the larger option for availability during rebuild; not included here. +Scope: default-off/quota-gated subsystem, bench tooling, and tests only; no delivery path changes. +Proof boundary: local CI-replica source/test evidence only; no commit, push, PR update, release, or runtime proof. diff --git a/adaptive_retrieval.py b/adaptive_retrieval.py index e3fd7cee6..21c668fb9 100644 --- a/adaptive_retrieval.py +++ b/adaptive_retrieval.py @@ -181,6 +181,10 @@ def parse(cls, value: Any) -> "EvidenceRequirement": "requirement description must contain " f"1..{MAX_REQUIREMENT_DESCRIPTION_CHARS} characters" ) + if any(0xD800 <= ord(char) <= 0xDFFF for char in description): + raise ValueError( + "requirement description must not contain surrogate code points" + ) raw_minimum = value.get("minimum_refs", 1) if isinstance(raw_minimum, bool): raise ValueError("minimum_refs must be an integer") diff --git a/benchmarking/h3_composition_replay.py b/benchmarking/h3_composition_replay.py index 1d344a1c8..e43132770 100644 --- a/benchmarking/h3_composition_replay.py +++ b/benchmarking/h3_composition_replay.py @@ -207,6 +207,21 @@ def global_top_state_ids(self, qid: str, limit: int = 128) -> set[int]: rows = store._fts_rows(expression, limit) return {int(row["state_id"]) for row in rows} + def fused_candidate_state_ids( + self, qid: str, **kwargs: Any + ) -> set[int]: + domain, text = self.questions[qid] + store = self.stores[domain] + store.injected = self._injected_ranks(qid) + store.query( + text, candidate_limit=128, limit=16, image_limit=0, + include_adjacent=True, text_char_limit=2000, **kwargs, + ) + return { + int(row["state_id"]) + for row in store.last_query_telemetry()["state_candidate_pool"] + } + def ref_to_state_id(self, qid: str, ref: str) -> int | None: match = _REF_RE.match(ref) if not match: @@ -275,13 +290,22 @@ def load_ground_truth(ctx: ReplayContext) -> dict[str, Any]: } -def measure_ceiling(ctx: ReplayContext, recovery_targets: dict[str, dict[str, Any]]) -> dict[str, Any]: - """Per vanished ref: is its state present in global_rows top-128?""" +def measure_ceiling( + ctx: ReplayContext, + recovery_targets: dict[str, dict[str, Any]], + *, + knob_kwargs: dict[str, Any] | None = None, +) -> dict[str, Any]: + """Per vanished ref: is its state present in the policy's candidate pool?""" per_ref: dict[str, dict[str, bool]] = {} + policy = knob_kwargs or {} for qid, info in recovery_targets.items(): if not info["vanished"]: continue - top = ctx.global_top_state_ids(qid, 128) + if "arm_quota" in policy: + top = ctx.fused_candidate_state_ids(qid, **policy) + else: + top = ctx.global_top_state_ids(qid, 128) for ref in info["vanished"]: state_id = ctx.ref_to_state_id(qid, ref) per_ref.setdefault(qid, {})[ref] = state_id is not None and state_id in top @@ -415,7 +439,14 @@ def main() -> int: results = [] for knob in knobs: - row = evaluate_knob(ctx, ground, ceiling, knob) + knob_ceiling = ( + measure_ceiling( + ctx, ground["recovery_targets"], knob_kwargs=knob + ) + if "arm_quota" in knob + else ceiling + ) + row = evaluate_knob(ctx, ground, knob_ceiling, knob) row["latency"] = measure_latency(ctx, sample, knob) row["latency_pct_vs_default"] = ( (row["latency"]["p95_ms"] / baseline_latency["p95_ms"] - 1.0) * 100.0 diff --git a/tests/test_adaptive_retrieval.py b/tests/test_adaptive_retrieval.py index 1ca4c7c72..6782fa410 100644 --- a/tests/test_adaptive_retrieval.py +++ b/tests/test_adaptive_retrieval.py @@ -285,6 +285,15 @@ def test_requirements_digest_distinguishes_descriptions(): assert requirements_digest([ceo]) != requirements_digest([cfo]) +def test_requirement_description_rejects_surrogate_code_points(): + with pytest.raises(ValueError, match="surrogate code points"): + EvidenceRequirement.parse({ + "slot_id": "role_holder", + "description": "the CEO of \ud800Acme", + "minimum_refs": 1, + }) + + def test_max_sized_requirements_validate_and_digest(tmp_path): requirements = [ { diff --git a/tests/test_h3_composition_replay.py b/tests/test_h3_composition_replay.py new file mode 100644 index 000000000..1a31d00e7 --- /dev/null +++ b/tests/test_h3_composition_replay.py @@ -0,0 +1,38 @@ +from __future__ import annotations + +from benchmarking.h3_composition_replay import measure_ceiling + + +class _CeilingContext: + def global_top_state_ids(self, qid, limit=128): + assert qid == "q1" + assert limit == 128 + return {1} + + def fused_candidate_state_ids(self, qid, **kwargs): + assert qid == "q1" + assert kwargs == {"arm_quota": (6, 5)} + return {1, 2} + + def ref_to_state_id(self, qid, ref): + assert qid == "q1" + return {"trajectory://test/a/state/0": 2}[ref] + + +def test_policy_d_ceiling_includes_scoped_fused_candidates(): + targets = { + "q1": { + "vanished": ["trajectory://test/a/state/0"], + "bucket": "composition", + } + } + + global_ceiling = measure_ceiling(_CeilingContext(), targets) + policy_d_ceiling = measure_ceiling( + _CeilingContext(), + targets, + knob_kwargs={"arm_quota": (6, 5)}, + ) + + assert global_ceiling["recoverable"] == 0 + assert policy_d_ceiling["recoverable"] == 1 diff --git a/tests/test_h5_state_semantic_replay.py b/tests/test_h5_state_semantic_replay.py index b46e567c8..c43f90c68 100644 --- a/tests/test_h5_state_semantic_replay.py +++ b/tests/test_h5_state_semantic_replay.py @@ -8,6 +8,15 @@ def test_output_parent_exists_before_provider_work(tmp_path, monkeypatch): output = tmp_path / "new" / "nested" / "sweep.json" monkeypatch.delenv("VOYAGE_API_KEY", raising=False) + + def fail_provider(): + raise AssertionError("provider must not be constructed without an API key") + + monkeypatch.setattr( + h5_state_semantic_replay, + "CachedVoyageQueryProvider", + fail_provider, + ) monkeypatch.setattr( sys, "argv", diff --git a/tests/test_trajectory_state_semantic_expansion.py b/tests/test_trajectory_state_semantic_expansion.py index fb0d52f83..1b25b45c4 100644 --- a/tests/test_trajectory_state_semantic_expansion.py +++ b/tests/test_trajectory_state_semantic_expansion.py @@ -262,6 +262,7 @@ def test_expansion_can_fill_a_pure_lexical_miss(tmp_path): def test_expansion_can_fill_a_termless_lexical_miss(tmp_path): store = _build_invisible_semantic_store(tmp_path) + provider = store.embedding_provider baseline = store.query("🚀", image_limit=0) baseline_telemetry = store.last_query_telemetry() @@ -269,6 +270,7 @@ def test_expansion_can_fill_a_termless_lexical_miss(tmp_path): assert explicit_off == baseline == () assert store.last_query_telemetry() == baseline_telemetry + query_calls_before = provider.query_calls expanded = store.query( "🚀", image_limit=0, @@ -278,6 +280,8 @@ def test_expansion_can_fill_a_termless_lexical_miss(tmp_path): assert len(expanded) == 1 assert expanded[0].match_kind == "state_semantic" + assert provider.query_calls == query_calls_before + 1 + assert store.last_query_telemetry()["source_candidate_ranks"] == [] def test_quota_caps_admissions(tmp_path): @@ -559,6 +563,53 @@ def test_interrupted_profile_rebuild_leaves_no_active_profile_until_cutover( ).fetchone()[0] == 1 +def test_forced_same_profile_rebuild_discards_prior_rows(tmp_path): + asset_root = tmp_path / "assets" + asset_root.mkdir() + initial = StateVectorProvider() + store = TrajectoryStore( + tmp_path / "lcm.db", _identity(), asset_root=asset_root, + embedding_provider=initial, + ) + store.insert(_source( + asset_root, + trajectory_id="answerpath", + ordinal=0, + goal="Update the account settings", + texts=( + "widget configuration export panel form", + "alpha-answer success banner", + "logout footer copyright notice", + ), + )) + store.finalize(["answerpath"]) + completed = store.build_state_semantic_index(initial) + + interrupted = InterruptingStateVectorProvider( + model_id=initial.model_id, fail_after=1 + ) + with pytest.raises(RuntimeError, match="interrupted backfill"): + store.build_state_semantic_index( + interrupted, + resume=False, + batch_max_items=1, + batch_token_budget=100_000, + ) + + assert store._conn.execute( + "SELECT COUNT(*) FROM lcm_trajectory_state_embeddings " + "WHERE profile_digest = ?", + (completed["profile_digest"],), + ).fetchone()[0] == 1 + + interrupted.fail_after = None + resumed = store.build_state_semantic_index( + interrupted, batch_max_items=1, batch_token_budget=100_000 + ) + assert resumed["already_embedded"] == 1 + assert resumed["states_embedded"] == 2 + + def test_dimension_probe_is_bounded_by_document_token_budget(tmp_path): asset_root = tmp_path / "assets" asset_root.mkdir() @@ -669,6 +720,38 @@ def test_oversize_chunks_pack_by_item_and_token_budgets(tmp_path): assert max(provider.request_token_counts) <= 9 +def test_smaller_batch_budget_routes_normal_document_through_chunks(tmp_path): + asset_root = tmp_path / "assets" + asset_root.mkdir() + provider = RequestBudgetProvider(token_limit=5) + store = TrajectoryStore( + tmp_path / "lcm.db", + _identity(), + asset_root=asset_root, + embedding_provider=provider, + ) + store.insert(_source( + asset_root, + trajectory_id="batch-budget", + ordinal=0, + goal="Honor the request cap", + texts=("alpha-answer " + ("token " * 20),), + )) + store.finalize(["batch-budget"]) + + stats = store.build_state_semantic_index( + provider, + document_token_budget=50, + batch_token_budget=5, + batch_max_items=4, + ) + + assert stats["states_embedded"] == 1 + assert stats["chunked_states"] == 1 + assert provider.request_token_counts + assert max(provider.request_token_counts) <= 5 + + def test_oversize_progress_can_stop_between_chunks_with_partial_spend(tmp_path): asset_root = tmp_path / "assets" asset_root.mkdir() diff --git a/trajectory_store.py b/trajectory_store.py index 930b92fc7..54b7a4266 100644 --- a/trajectory_store.py +++ b/trajectory_store.py @@ -1724,6 +1724,12 @@ def build_state_semantic_index( now, ), ) + if not resume: + self._conn.execute( + "DELETE FROM lcm_trajectory_state_embeddings " + "WHERE profile_digest = ?", + (profile_digest,), + ) self._conn.commit() except Exception: self._conn.rollback() @@ -1762,6 +1768,9 @@ def _emit_progress() -> None: # Partition pending states into single-request documents and the # oversize (chunked) minority, both packed to the same item/token caps. + request_document_token_budget = min( + document_token_budget, batch_token_budget + ) normal: list[tuple[int, str, str, int]] = [] # (state_id, doc, sha, tokens) oversize: list[tuple[int, str, str]] = [] # (state_id, doc, sha) for row in pending: @@ -1769,7 +1778,7 @@ def _emit_progress() -> None: document = self._state_embed_document(row["text"], row["url"], state_id) document_sha = _sha256_text(document) tokens = count_tokens(document) - if tokens > document_token_budget: + if tokens > request_document_token_budget: oversize.append((state_id, document, document_sha)) else: normal.append((state_id, document, document_sha, tokens)) @@ -1840,7 +1849,9 @@ def _flush_normal() -> None: # --- oversize (chunked) states ---------------------------------------- for state_id, document, document_sha in oversize: - chunks = self._state_token_chunks(document, document_token_budget) + chunks = self._state_token_chunks( + document, request_document_token_budget + ) chunk_vectors: list[tuple[float, ...]] = [] start = 0 while start < len(chunks): @@ -3259,41 +3270,42 @@ def query( ) semantic_ranks: list[tuple[int, float]] = [] semantic_attempt: TrajectorySemanticAttempt | None = None - attempt_started = time.monotonic() - calls_before = self._semantic_usage["query_calls"] - try: - semantic_ranks = self._semantic_source_ranks(query) - except Exception as exc: - # Restore historical semantics FIRST, unconditionally, before any - # introspection can fail: the fallback counter must bump even if the - # (hostile) exception explodes during telemetry recording. - self._semantic_usage["fallbacks"] += 1 - fallback_latency_ms = (time.monotonic() - attempt_started) * 1000.0 + if expression: + attempt_started = time.monotonic() + calls_before = self._semantic_usage["query_calls"] try: - # Was a bare ``except Exception: fallbacks += 1`` that discarded - # the failure class/status. Now the typed reason survives -- and - # the whole record step is itself fenced so an exotic exception - # (kind/status_code/retry_after as raising properties) degrades - # to a clean FTS fallback instead of failing the query. - semantic_attempt = self._record_semantic_attempt( - outcome="fallback", - latency_ms=fallback_latency_ms, - exception=exc, - ) - except Exception: - semantic_attempt = self._record_minimal_fallback_attempt( - latency_ms=fallback_latency_ms, - ) - else: - # Only record a success when an embed was actually dispatched; an - # early return (no provider / profile mismatch) is a skip, not an - # attempt, and must not inflate the success count. - if self._semantic_usage["query_calls"] > calls_before: - semantic_attempt = self._record_semantic_attempt( - outcome="success", - latency_ms=(time.monotonic() - attempt_started) * 1000.0, - exception=None, - ) + semantic_ranks = self._semantic_source_ranks(query) + except Exception as exc: + # Restore historical semantics FIRST, unconditionally, before any + # introspection can fail: the fallback counter must bump even if the + # (hostile) exception explodes during telemetry recording. + self._semantic_usage["fallbacks"] += 1 + fallback_latency_ms = (time.monotonic() - attempt_started) * 1000.0 + try: + # Was a bare ``except Exception: fallbacks += 1`` that discarded + # the failure class/status. Now the typed reason survives -- and + # the whole record step is itself fenced so an exotic exception + # (kind/status_code/retry_after as raising properties) degrades + # to a clean FTS fallback instead of failing the query. + semantic_attempt = self._record_semantic_attempt( + outcome="fallback", + latency_ms=fallback_latency_ms, + exception=exc, + ) + except Exception: + semantic_attempt = self._record_minimal_fallback_attempt( + latency_ms=fallback_latency_ms, + ) + else: + # Only record a success when an embed was actually dispatched; an + # early return (no provider / profile mismatch) is a skip, not an + # attempt, and must not inflate the success count. + if self._semantic_usage["query_calls"] > calls_before: + semantic_attempt = self._record_semantic_attempt( + outcome="success", + latency_ms=(time.monotonic() - attempt_started) * 1000.0, + exception=None, + ) scoped_rows = self._fts_rows( expression, candidate_limit,