Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
50 changes: 47 additions & 3 deletions tests/bench_gate/test_vocab_bridge_uplift.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,19 +22,26 @@
{
"id": "row-id",
"query": "raw query string passed to retrieve",
"k": 10, # optional, default 10
"store_beliefs": [
{"id": "b1", "content": "...", "anchors": ["text", "..."]},
...
],
"expected_canonicals": ["sqlite", "python", "..."]
"expected_canonicals": ["sqlite", "python", "..."],
"expected_top_k": ["b1", "b2", ...] # optional; A2 ship gate
}

`store_beliefs[i].anchors` is optional; when present, each anchor
string is added as an inbound edge from a synthetic citing belief
to seed the bridge with anchor-source surface forms (#148 parity).
`expected_canonicals` is the set of canonical-entity tokens that
the bridge SHOULD append for this query; the test asserts at least
one is present in the rewritten output.
the bridge SHOULD append for this query; the precondition test
asserts at least one is present in the rewritten output.
`expected_top_k` is the ground-truth belief-id ranking; the strict
A2 ship gate (``test_vocab_bridge_ship_gate_runner_present``) runs
``retrieve_v2`` with the bridge OFF then ON and asserts NDCG@k goes
strictly up. Rows without ``expected_top_k`` are skipped by the
ship gate but still exercised by the precondition gate.
"""
from __future__ import annotations

Expand Down Expand Up @@ -133,3 +140,40 @@ def test_bridge_appends_at_least_one_expected_canonical(
f"{n_hits}/{n_rows} rows ({coverage:.1%}); harvest/rewrite "
f"regression suspected (was the surface-form pipeline broken?)"
)


@pytest.mark.bench_gated
def test_vocab_bridge_ship_gate_runner_present(
aelfrice_corpus_root: Path,
) -> None:
"""The full A2 NDCG@k ship gate runs from
``tests.retrieve_uplift_runner.run_vocab_bridge_uplift``. This test
skips when the runner is absent or when no row carries
``expected_top_k`` — the runner is the operator-side gate for
flipping ``use_vocab_bridge`` to default-on (#433 Phase 2).
"""
rows = load_corpus_module(aelfrice_corpus_root, "vocab_bridge")
assert rows, "vocab_bridge corpus produced zero rows"

Comment on lines +155 to +157

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue | 🟠 Major | ⚡ Quick win

Skip on empty corpus instead of failing the bench gate

Line 156 hard-fails when the corpus is absent/empty, which makes this gate brittle in environments without lab resources. This should skip like the other absent-resource paths.

Proposed fix
     rows = load_corpus_module(aelfrice_corpus_root, "vocab_bridge")
-    assert rows, "vocab_bridge corpus produced zero rows"
+    if not rows:
+        pytest.skip(
+            "vocab_bridge corpus produced zero rows; "
+            "add rows under tests/corpus/v2_0/vocab_bridge/*.jsonl "
+            "before enabling this ship gate"
+        )
📝 Committable suggestion

‼️ IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
rows = load_corpus_module(aelfrice_corpus_root, "vocab_bridge")
assert rows, "vocab_bridge corpus produced zero rows"
rows = load_corpus_module(aelfrice_corpus_root, "vocab_bridge")
if not rows:
pytest.skip(
"vocab_bridge corpus produced zero rows; "
"add rows under tests/corpus/v2_0/vocab_bridge/*.jsonl "
"before enabling this ship gate"
)
🤖 Prompt for AI Agents
Verify each finding against current code. Fix only still-valid issues, skip the
rest with a brief reason, keep changes minimal, and validate.

In `@tests/bench_gate/test_vocab_bridge_uplift.py` around lines 155 - 157, The
test currently hard-fails when load_corpus_module(aelfrice_corpus_root,
"vocab_bridge") returns no rows; change that behavior to skip the test instead.
Replace the assert rows check with a conditional that calls pytest.skip(...)
with a clear reason when rows is empty or falsy (and add an import for pytest if
not already present). Keep the rest of the test using the loaded rows when
present.

runner_mod = pytest.importorskip(
"tests.retrieve_uplift_runner",
reason=(
"vocab_bridge uplift runner not yet wired "
"(operator gate; spec § A2 — pending lab-side corpus + scorer)"
),
)

if not any(r.get("expected_top_k") for r in rows):
pytest.skip(
"vocab_bridge corpus has no rows with expected_top_k; "
"annotate at least one row before this gate can fire"
)

results = runner_mod.run_vocab_bridge_uplift(rows)
assert results.uplift > 0, (
"use_vocab_bridge must show strictly positive NDCG@k uplift\n"
f" ON={results.mean_ndcg_on:.4f} "
f"OFF={results.mean_ndcg_off:.4f} "
f"uplift={results.uplift:+.4f} "
f"n_rows={results.n_rows}"
)
174 changes: 174 additions & 0 deletions tests/retrieve_uplift_runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -609,6 +609,180 @@ def run_query_strategy_uplift(
)


# ---------------------------------------------------------------------
# Vocabulary bridge (#433) — use_vocab_bridge bench gate (spec § A2).
#
# Consumed by tests/bench_gate/test_vocab_bridge_uplift.py.
# Row schema (`tests/corpus/v2_0/vocab_bridge/*.jsonl`) — extends the
# precondition row schema with `expected_top_k` and optional `k`:
#
# {
# "id": "row-id",
# "query": "raw query string",
# "k": 10,
# "store_beliefs": [
# {"id": "b1", "content": "...", "anchors": ["text", "..."]},
# ...
# ],
# "expected_canonicals": ["sqlite", ...], # used by precondition
# "expected_top_k": ["b1", "b2", ...] # ground-truth ranking
# }
#
# `expected_canonicals` stays the precondition surface — it lets the
# appends-at-least-one gate fire without naming specific belief ids.
# `expected_top_k` is the strict-NDCG ground truth: the bridge wins
# only if rewriting the query surfaces these ids in this order vs the
# default-OFF baseline.
# ---------------------------------------------------------------------

_BENCH_TS = "2026-05-08T00:00:00Z"


@dataclass(frozen=True)
class VocabBridgeUplift:
"""#433 result shape.

Field names mirror ``DocLinkerUpliftResults`` / ``QueryStrategyUplift``
so the bench-gate failure-message formatter reads
``mean_ndcg_off`` / ``mean_ndcg_on`` / ``uplift`` without
per-runner branching.
"""

n_rows: int
mean_ndcg_off: float
mean_ndcg_on: float

@property
def uplift(self) -> float:
return self.mean_ndcg_on - self.mean_ndcg_off


def _seed_store_for_vocab_bridge(
store: MemoryStore, row: dict, # type: ignore[type-arg]
) -> None:
"""Seed a store from the vocab_bridge row shape.

Row carries ``store_beliefs`` (not ``beliefs``) where each entry
optionally lists ``anchors`` — surface-form strings written as
incoming-edge ``anchor_text`` from synthetic citing beliefs. This
mirrors the precondition seeder in
``tests/bench_gate/test_vocab_bridge_uplift.py`` so the runner
sees an identical store shape; the bridge harvest/rewrite path
then operates on the same anchor-text universe under both arms.
"""
for entry in row.get("store_beliefs", []):
bid = str(entry["id"])
store.insert_belief(
Belief(
id=bid,
content=str(entry["content"]),
content_hash=f"corpus:{bid}",
alpha=float(entry.get("alpha", 1.0)),
beta=float(entry.get("beta", 1.0)),
type=entry.get("type", BELIEF_FACTUAL),
lock_level=LOCK_NONE,
locked_at=None,
demotion_pressure=0,
created_at=_BENCH_TS,
last_retrieved_at=None,
origin=ORIGIN_AGENT_INFERRED,
)
)
for entry in row.get("store_beliefs", []):
for j, anchor in enumerate(entry.get("anchors") or []):
citer_id = f"_a_{entry['id']}_{j}"
if not store.get_belief(citer_id):
store.insert_belief(
Belief(
id=citer_id,
content="anchor citer",
content_hash=f"corpus:{citer_id}",
alpha=1.0,
beta=1.0,
type=BELIEF_FACTUAL,
lock_level=LOCK_NONE,
locked_at=None,
demotion_pressure=0,
created_at=_BENCH_TS,
last_retrieved_at=None,
origin=ORIGIN_AGENT_INFERRED,
)
)
store.insert_edge(
Edge(
src=citer_id,
dst=str(entry["id"]),
type="CITES",
weight=1.0,
anchor_text=str(anchor),
)
)


def run_vocab_bridge_uplift(
rows: list[dict], # type: ignore[type-arg]
) -> VocabBridgeUplift:
"""Spec § A2 vocabulary-bridge uplift driver.

Per row, runs ``retrieve_v2`` twice on fresh stores seeded from
``store_beliefs`` (with anchors): once with
``use_vocab_bridge=False`` (baseline) and once with ``=True``.
Same query, same store shape, same ``k``. NDCG@k is scored against
``expected_top_k`` and averaged across rows.

Rows missing ``expected_top_k`` are skipped (the precondition gate
only needs ``expected_canonicals``). The shipped substrate is
default-OFF: the bridge appends canonical-entity tokens to the
query, never substitutes — so on a corpus where BM25 alone
already ranks the relevant beliefs above noise, ``uplift`` will
be ~0 and the strict ``> 0`` ship-gate assertion will fail. That
is the correct gate-broken signal until labelled rows expose
surface-form gaps the bridge can close.
"""
scoreable = [r for r in rows if r.get("expected_top_k")]
n = len(scoreable)
if n == 0:
return VocabBridgeUplift(0, 0.0, 0.0)

off_total = 0.0
on_total = 0.0
with tempfile.TemporaryDirectory() as tmp:
tmp_root = Path(tmp)
for row in scoreable:
k = _default_k(row)
expected = list(row["expected_top_k"])

for bridge_on in (False, True):
_db_counter[0] += 1
db = (
tmp_root
/ f"vb_{row['id']}_{int(bridge_on)}_{_db_counter[0]}.db"
)
store = MemoryStore(str(db))
try:
_seed_store_for_vocab_bridge(store, row)
result = retrieve_v2(
store, row["query"],
budget=DEFAULT_TOKEN_BUDGET,
use_entity_index=False,
use_vocab_bridge=bridge_on,
)
returned = [b.id for b in result.beliefs[:k]]
finally:
store.close()
ndcg = ndcg_at_k(returned, expected, k)
if bridge_on:
on_total += ndcg
else:
off_total += ndcg

return VocabBridgeUplift(
n_rows=n,
mean_ndcg_off=off_total / n,
mean_ndcg_on=on_total / n,
)


def _format_table(results: list[FlagUplift]) -> str:
lines = [
f"{'flag':<28} {'n':>4} {'NDCG_off':>10} {'NDCG_on':>10} {'uplift':>10}",
Expand Down
67 changes: 67 additions & 0 deletions tests/test_retrieve_uplift_runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -282,6 +282,73 @@ def test_query_strategy_uplift_lowercase_no_extremes_means_zero_uplift() -> None
assert r.uplift == 0.0


def test_vocab_bridge_uplift_empty_input() -> None:
from tests.retrieve_uplift_runner import run_vocab_bridge_uplift

r = run_vocab_bridge_uplift([])
assert r.n_rows == 0
assert r.mean_ndcg_off == 0.0
assert r.mean_ndcg_on == 0.0
assert r.uplift == 0.0


def test_vocab_bridge_uplift_skips_rows_without_expected_top_k() -> None:
"""Precondition rows (those carrying only ``expected_canonicals``)
have no ground-truth ranking, so they must be skipped — not silently
scored as zero, which would dilute the mean and falsely shrink the
uplift signal. ``n_rows`` reflects the scoreable subset."""
from tests.retrieve_uplift_runner import run_vocab_bridge_uplift

rows = [
{
"id": "vb-precondition-only",
"query": "alpha",
"store_beliefs": [{"id": "a", "content": "alpha alpha"}],
"expected_canonicals": ["alpha"],
},
]
r = run_vocab_bridge_uplift(rows)
assert r.n_rows == 0
assert r.uplift == 0.0


def test_vocab_bridge_uplift_runs_on_synthetic_row() -> None:
"""OFF/ON arms run end-to-end without raising; metrics are bounded.

Real uplift on a synthetic row is unlikely (the bridge harvests
canonical-entity superpositions from corpus surface forms; a
two-belief store doesn't expose a meaningful canonical universe).
The contract under test is shape + bounded metric, not that the
rewrite influences ranking — the corpus uplift is the lab-side
gate."""
from tests.retrieve_uplift_runner import run_vocab_bridge_uplift

row = {
"id": "vb-test-001",
"query": "memory store",
"k": 3,
"store_beliefs": [
{
"id": "b1",
"content": "the memory store persists beliefs",
"anchors": ["MemoryStore"],
},
{"id": "b2", "content": "the configuration file lives at /etc"},
{
"id": "b3",
"content": "the memory store uses sqlite",
"anchors": ["sqlite"],
},
],
"expected_canonicals": ["sqlite"],
"expected_top_k": ["b1", "b3"],
}
r = run_vocab_bridge_uplift([row])
assert r.n_rows == 1
assert 0.0 <= r.mean_ndcg_off <= 1.0
assert 0.0 <= r.mean_ndcg_on <= 1.0


def test_run_per_flag_uplift_covers_all_flags() -> None:
"""Hypothesis: the harness reports one row per registered flag.
Falsifiable if a flag is silently dropped."""
Expand Down
Loading