Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
110 changes: 76 additions & 34 deletions benchmarks/results/v2.0.0.json
Original file line number Diff line number Diff line change
@@ -1,8 +1,8 @@
{
"_calibration_notes": "First canonical pass 2026-05-07. 6 of 11 invocations succeeded (MAB \u00d74, LongMemEval, AMA-Bench); 5 failed due to missing /tmp/ data dirs (LoCoMo + StructMemEval \u00d74). Re-run those adapters after populating /tmp/LoCoMo and /tmp/StructMemEval. Per-metric override bands (3+ runs \u00d7 1.5 spread) not yet calibrated \u2014 first-run values are placeholders for the working adapters.",
"aelfrice_version": "1.6.0",
"captured_at_utc": "2026-05-07T02:10:35Z",
"git_commit": "bce8311",
"_calibration_notes": "Calibrated 2026-05-08 from 3 full canonical runs (11/11 ok each) on git_commit 9e3f8dee. Tolerance bands per spec: observed range x 1.5; metric_overrides left empty (defaults \u2014 F1 plus or minus 7%, EM plus or minus 10%, latency plus or minus 25%, fallback plus or minus 10%, absolute floor plus or minus 0.02 \u2014 cover observed variance. Across 3 runs, MAB and StructMemEval metrics are bit-deterministic; LongMemEval avg_latency_ms varied 5.55\u20136.04 ms (well inside default plus or minus 25%). Per-run inputs held in /tmp/v2.0.0-cal/run-{1,2,3}.json during calibration session.",
"aelfrice_version": "2.0.0",
"captured_at_utc": "2026-05-08T15:00:00Z",
"git_commit": "9e3f8dee2898d21ed5250618d86f84b3235caab4",
"harness_version": "1",
"headline_cut": {
"amabench": [
Expand Down Expand Up @@ -92,12 +92,12 @@
}
]
},
"label": "v2.0.0 canonical (first calibration pass \u2014 partial)",
"label": "v2.0.0 canonical",
"metric_overrides": {},
"results": {
"amabench": {
"_": {
"_elapsed_sec": 266.921,
"_elapsed_sec": 189.73,
"_status": "ok",
"output": {
"domain_counts": {
Expand All @@ -121,59 +121,69 @@
},
"locomo": {
"_": {
"_elapsed_sec": 1.041,
"_error_message": "Traceback (most recent call last):\n File \"<frozen runpy>\", line 198, in _run_module_as_main\n File \"<frozen runpy>\", line 88, in _run_code\n File \"$HOME/projects/aelfrice/benchmarks/locomo_adapter.py\", line 565, in <module>\n main()\n ~~~~^^\n File \"$HOME/projects/aelfrice/benchmarks/locomo_adapter.py\", line 503, in main\n conversations: list[LoCoMoConversation] = load_locomo(args.data)\n ~~~~~~~~~~~^^^^^^^^^^^\n File \"/Users",
"_status": "error"
"_elapsed_sec": 34.403,
"_status": "ok",
"output": {
"category_f1": {
"1": 0.1051,
"2": 0.0074,
"3": 0.008,
"4": 0.011,
"5": 0.0
},
"overall_f1": 0.0212,
"total_qa": 1986
}
}
},
"longmemeval": {
"_": {
"_elapsed_sec": 70.552,
"_elapsed_sec": 51.822,
"_status": "ok",
"output": {
"avg_beliefs_per_query": 49.15,
"avg_latency_ms": 7.8,
"avg_latency_ms": 5.55,
"category_stats": {
"knowledge-update": {
"avg_beliefs": 50.45,
"avg_latency_ms": 8.0,
"avg_latency_ms": 6.05,
"count": 78
},
"multi-session": {
"avg_beliefs": 50.51,
"avg_latency_ms": 10.43,
"avg_latency_ms": 7.19,
"count": 133
},
"single-session-assistant": {
"avg_beliefs": 38.93,
"avg_latency_ms": 2.9,
"avg_latency_ms": 2.06,
"count": 56
},
"single-session-preference": {
"avg_beliefs": 50.23,
"avg_latency_ms": 5.4,
"avg_latency_ms": 3.77,
"count": 30
},
"single-session-user": {
"avg_beliefs": 49.94,
"avg_latency_ms": 4.9,
"avg_latency_ms": 3.37,
"count": 70
},
"temporal-reasoning": {
"avg_beliefs": 50.68,
"avg_latency_ms": 9.16,
"avg_latency_ms": 6.64,
"count": 133
}
},
"total_ingest_time_s": 61.63,
"total_ingest_time_s": 44.69,
"total_ingest_turns": 10960,
"total_questions": 500
}
}
},
"mab": {
"Accurate_Retrieval": {
"_elapsed_sec": 541.933,
"_elapsed_sec": 381.34,
"_status": "ok",
"output": {
"exact_match": 0.0,
Expand All @@ -185,7 +195,7 @@
}
},
"Conflict_Resolution": {
"_elapsed_sec": 83.214,
"_elapsed_sec": 60.041,
"_status": "ok",
"output": {
"exact_match": 0.0,
Expand All @@ -197,7 +207,7 @@
}
},
"Long_Range_Understanding": {
"_elapsed_sec": 540.45,
"_elapsed_sec": 385.399,
"_status": "ok",
"output": {
"exact_match": 0.0,
Expand All @@ -209,7 +219,7 @@
}
},
"Test_Time_Learning": {
"_elapsed_sec": 452.277,
"_elapsed_sec": 311.118,
"_status": "ok",
"output": {
"exact_match": 0.0,
Expand All @@ -223,24 +233,56 @@
},
"structmemeval": {
"accounting": {
"_elapsed_sec": 0.266,
"_error_message": "adapter exited 0 but did not write $TMPDIR/T/structmemeval_accounting.json",
"_status": "error"
"_elapsed_sec": 0.658,
"_status": "ok",
"output": {
"accuracy": 0.0,
"bench": "big",
"perfect_cases": 0,
"task": "accounting",
"total_cases": 15,
"total_correct": 0,
"total_queries": 15
}
},
"location": {
"_elapsed_sec": 0.406,
"_error_message": "adapter exited 0 but did not write $TMPDIR/T/structmemeval_location.json",
"_status": "error"
"_elapsed_sec": 1.526,
"_status": "ok",
"output": {
"accuracy": 0.9048,
"bench": "big",
"perfect_cases": 38,
"task": "location",
"total_cases": 42,
"total_correct": 38,
"total_queries": 42
}
},
"recommendations": {
"_elapsed_sec": 0.271,
"_error_message": "adapter exited 0 but did not write $TMPDIR/T/structmemeval_recommendations.json",
"_status": "error"
"_elapsed_sec": 14.803,
"_status": "ok",
"output": {
"accuracy": 0.154,
"bench": "big",
"perfect_cases": 0,
"task": "recommendations",
"total_cases": 84,
"total_correct": 170,
"total_queries": 1104
}
},
"tree": {
"_elapsed_sec": 0.266,
"_error_message": "adapter exited 0 but did not write $TMPDIR/T/structmemeval_tree.json",
"_status": "error"
"_elapsed_sec": 2.141,
"_status": "ok",
"output": {
"accuracy": 0.0,
"bench": "big",
"perfect_cases": 0,
"task": "tree",
"total_cases": 22,
"total_correct": 0,
"total_queries": 43
}
}
}
},
Expand Down
1 change: 1 addition & 0 deletions benchmarks/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -195,6 +195,7 @@ def _default_runner(cmd: list[str], out_path: Path) -> subprocess.CompletedProce
# needs to read it directly.
_DETAIL_FIELDS_TO_STRIP: frozenset[str] = frozenset({
"per_question",
"per_case",
})


Expand Down
9 changes: 7 additions & 2 deletions tests/test_bench_dispatcher.py
Original file line number Diff line number Diff line change
Expand Up @@ -179,13 +179,16 @@ def test_headline_cut_recorded(tmp_path):


def test_per_question_detail_stripped_from_merged_output(tmp_path):
"""The merged JSON drops `per_question` lists — they bloat the file
and aren't read by the band-check (#437 calibration finding 2026-05-06).
"""The merged JSON drops `per_question` and `per_case` lists — they bloat
the file and aren't read by the band-check (#437 calibration finding
2026-05-06; `per_case` added 2026-05-08 after structmemeval bloated the
canonical to 6.4 MB).
"""
out = tmp_path / "stripped.json"
payload = {
"f1": 0.5, "exact_match": 0.3,
"per_question": [{"id": i, "score": 0.5} for i in range(2000)],
"per_case": [{"case_id": f"c{i}", "accuracy": 1.0} for i in range(50)],
}
rc = bench_run.main_all(
out_path=out, canonical=False, smoke=True,
Expand All @@ -199,6 +202,8 @@ def test_per_question_detail_stripped_from_merged_output(tmp_path):
ama_out = data["results"]["amabench"]["_"]["output"]
assert "per_question" not in mab_out
assert "per_question" not in ama_out
assert "per_case" not in mab_out
assert "per_case" not in ama_out
# Summary metrics retained.
assert mab_out["f1"] == 0.5
assert ama_out["exact_match"] == 0.3
Expand Down
15 changes: 12 additions & 3 deletions tests/test_benchmarks_badge.py
Original file line number Diff line number Diff line change
Expand Up @@ -90,13 +90,22 @@ def test_skipped_counts_as_not_ok(tmp_path):
assert text.startswith("reproducibility: ⚠️")


def test_canonical_v200_partial(tmp_path):
"""Sanity: today's checked-in canonical reports 6/11."""
def test_canonical_v200_full_pass(tmp_path):
"""Sanity: today's checked-in canonical reports 11/11 ok.

Was 6/11 during the partial first-pass calibration on 2026-05-07
(LoCoMo data missing + StructMemEval × 4 hitting the #473
temporal_sort kwarg bug). Calibrated to 11/11 on 2026-05-08 once
#473 shipped and /tmp/LoCoMo + /tmp/StructMemEval data dirs were
populated. Ratchets to detect regression — flip back to a partial
cut would surface here.
"""
canonical = Path(__file__).parent.parent / "benchmarks" / "results" / "v2.0.0.json"
if not canonical.exists():
pytest.skip("canonical baseline not present")
text = badge.compute_badge_text(canonical, today="2026-05-08")
assert "6/11 ok" in text
assert "11/11 ok" in text
assert text.startswith("reproducibility: ✅")

Comment on lines +93 to 109

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue | 🟡 Minor | ⚡ Quick win

Replace ambiguous × (MULTIPLICATION SIGN) with x in the docstring.

Ruff RUF002 flags line 97: StructMemEval × 4 uses U+00D7 (×) which is visually ambiguous with the Latin letter x. A simple substitution silences the lint warning.

🔧 Proposed fix
-    (LoCoMo data missing + StructMemEval × 4 hitting the `#473`
+    (LoCoMo data missing + StructMemEval x 4 hitting the `#473`
📝 Committable suggestion

‼️ IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
def test_canonical_v200_full_pass(tmp_path):
"""Sanity: today's checked-in canonical reports 11/11 ok.
Was 6/11 during the partial first-pass calibration on 2026-05-07
(LoCoMo data missing + StructMemEval × 4 hitting the #473
temporal_sort kwarg bug). Calibrated to 11/11 on 2026-05-08 once
#473 shipped and /tmp/LoCoMo + /tmp/StructMemEval data dirs were
populated. Ratchets to detect regressionflip back to a partial
cut would surface here.
"""
canonical = Path(__file__).parent.parent / "benchmarks" / "results" / "v2.0.0.json"
if not canonical.exists():
pytest.skip("canonical baseline not present")
text = badge.compute_badge_text(canonical, today="2026-05-08")
assert "6/11 ok" in text
assert "11/11 ok" in text
assert text.startswith("reproducibility: ✅")
def test_canonical_v200_full_pass(tmp_path):
"""Sanity: today's checked-in canonical reports 11/11 ok.
Was 6/11 during the partial first-pass calibration on 2026-05-07
(LoCoMo data missing + StructMemEval x 4 hitting the `#473`
temporal_sort kwarg bug). Calibrated to 11/11 on 2026-05-08 once
`#473` shipped and /tmp/LoCoMo + /tmp/StructMemEval data dirs were
populated. Ratchets to detect regressionflip back to a partial
cut would surface here.
"""
canonical = Path(__file__).parent.parent / "benchmarks" / "results" / "v2.0.0.json"
if not canonical.exists():
pytest.skip("canonical baseline not present")
text = badge.compute_badge_text(canonical, today="2026-05-08")
assert "11/11 ok" in text
assert text.startswith("reproducibility: ✅")
🧰 Tools
🪛 Ruff (0.15.12)

[warning] 97-97: Docstring contains ambiguous × (MULTIPLICATION SIGN). Did you mean x (LATIN SMALL LETTER X)?

(RUF002)

🤖 Prompt for AI Agents
Verify each finding against current code. Fix only still-valid issues, skip the
rest with a brief reason, keep changes minimal, and validate.

In `@tests/test_benchmarks_badge.py` around lines 93 - 109, In the
test_canonical_v200_full_pass docstring replace the Unicode multiplication sign
"×" (U+00D7) with a plain ASCII "x" to satisfy Ruff RUF002; locate the
triple-quoted string inside the test_canonical_v200_full_pass function and edit
the phrase "StructMemEval × 4" to "StructMemEval x 4" (no other changes
required).


def test_zero_total_does_not_render_check(tmp_path):
Expand Down
Loading