From b7c91de7dcf4cb2750190c340a88544d3fbb1ead Mon Sep 17 00:00:00 2001 From: pxtroniwnl Date: Sat, 26 Sep 2026 01:40:04 -0500 Subject: [PATCH] Restore the shape777 report schema the committed evidence and verifier read benchmarks/shape777.py no longer produced a report verify_published.py can consume. It emitted version shape777-published-v1, modes fresh and parallel_shared, and decisions_per_second, while results/raw/shape777-direct.json and the verifier at verify_published.py:89-96 use shape777-direct-v1, modes fresh_batch1 and parallel_suffix, and judgments_per_second. The verifier zips summary claims against raw results positionally, so a fresh run no longer lined up with the claims it is meant to reproduce. The rename also dropped evidence fields rather than renaming them: states, judgments, the p95 latency, cache_hits, padded_suffix_tokens and true_suffix_tokens were lost, as were passed, mean_probability_difference, tolerance and the non-gating note in comparisons_to_fresh, plus the top-level max_tokens and timing_scope. shape777_reranker.py had drifted the same way, losing max_tokens, timing_scope, states and yes_no_pair_forwards and reporting version shape777-reranker-published-v1. Restore both scripts to the published v1 contract and declare it as module constants so the schema is checkable. The report labels the unbatched mode fresh_batch1 while its committed prediction rows say fresh, so the two vocabularies stay distinct instead of being unified. comparisons_to_fresh.passed cannot be reconstructed from the committed data, where it is false under either reading, so it is defined as exact reproduction of fresh, which is the only interpretation consistent with the note stating the tolerance is non-gating. mean_probability_difference is the mean over option cells rather than over decisions; both were confirmed against the committed row-level predictions. tests/test_benchmark_schema.py pins the committed reports to those constants and fails if the scripts drift again, and checks that the keys verify_published.py reads are still emitted. All six tests fail against the previous scripts. No committed evidence is touched: results/raw, results/raw/SHA256SUMS and results/phase1-summary.json are unchanged, and verify_published.py still reproduces all 69 claims. --- benchmarks/shape777.py | 134 +++++++++++++++++++++++++------- benchmarks/shape777_reranker.py | 36 ++++++++- tests/test_benchmark_schema.py | 80 +++++++++++++++++++ 3 files changed, 223 insertions(+), 27 deletions(-) create mode 100644 tests/test_benchmark_schema.py diff --git a/benchmarks/shape777.py b/benchmarks/shape777.py index feeb301..44c7867 100644 --- a/benchmarks/shape777.py +++ b/benchmarks/shape777.py @@ -15,6 +15,65 @@ from semif_phase1.serial import SerialPrefixScorer from semif_phase1.shared import score_shared +REPORT_VERSION = "shape777-direct-v1" +TIMING_SCOPE = ( + "Warm loaded model; complete mode wall time includes tokenization, H2D, forward " + "and D2H; excludes model load and output write." +) +COMPARISON_TOLERANCE = 1.0 +COMPARISON_NOTE = ( + "Tolerance is deliberately non-gating; report exact probability drift and all " + "argmax flips." +) +COMMON_FIELDS = ( + "judgments", + "states", + "wall_seconds", + "judgments_per_second", + "state_latency_p50_seconds", + "state_latency_p95_seconds", + "peak_cuda_bytes", +) +RESULT_FIELDS = { + "fresh_batch1": COMMON_FIELDS, + "serial_prefix": (*COMMON_FIELDS, "cache_hits"), + "parallel_suffix": (*COMMON_FIELDS, "padded_suffix_tokens", "true_suffix_tokens"), +} +COMPARISON_FIELDS = ( + "passed", + "max_probability_difference", + "mean_probability_difference", + "argmax_flips", + "tolerance", +) +# The report labels the unbatched mode as fresh_batch1 while its prediction rows +# were committed as fresh, so the two vocabularies are kept distinct. +MODES = ("fresh_batch1", "serial_prefix", "parallel_suffix") +PREDICTION_LABELS = { + "fresh_batch1": "fresh", + "serial_prefix": "serial_prefix", + "parallel_suffix": "parallel_suffix", +} +TOP_LEVEL_FIELDS = ( + "version", + "input_sha256", + "model", + "hardware", + "max_tokens", + "timing_scope", + "results", + "comparisons_to_fresh", +) + + +def percentile(values, fraction): + ordered = sorted(values) + index = (len(ordered) - 1) * fraction + lower = int(index) + return ordered[lower] + ( + ordered[min(lower + 1, len(ordered) - 1)] - ordered[lower] + ) * (index - lower) + def main() -> None: parser = argparse.ArgumentParser(description=__doc__) @@ -42,64 +101,87 @@ def main() -> None: warm_serial.score(row) score_shared(model, tokenizer, first, metadata, args.max_tokens) report = { - "version": "shape777-published-v1", + "version": REPORT_VERSION, "input_sha256": hashlib.sha256(args.input.read_bytes()).hexdigest(), "model": metadata, "hardware": torch.cuda.get_device_name(0), - "timing_scope": "Warm model; includes prompt construction, tokenization, transfers, forward passes and CPU readout.", + "max_tokens": args.max_tokens, + "timing_scope": TIMING_SCOPE, "results": [], } predictions = {} - for mode in ("fresh", "serial_prefix", "parallel_shared"): + for mode in MODES: torch.cuda.reset_peak_memory_stats() started = time.perf_counter() values, state_times = [], [] + cache_hits = 0 + padded_suffix_tokens = 0 + true_suffix_tokens = 0 for group in groups.values(): mark = time.perf_counter() - if mode == "fresh": + if mode == "fresh_batch1": values.extend(score(model, tokenizer, row, metadata, args.max_tokens) for row in group) elif mode == "serial_prefix": scorer = SerialPrefixScorer(model, tokenizer, metadata, args.max_tokens) - values.extend(scorer.score(row) for row in group) + for row in group: + result = scorer.score(row) + cache_hits += bool(result["cache_hit"]) + values.append(result) else: - scored, _ = score_shared(model, tokenizer, group, metadata, args.max_tokens) + scored, timing = score_shared(model, tokenizer, group, metadata, args.max_tokens) values.extend(scored) + padded_suffix_tokens += timing["padded_suffix_tokens"] + true_suffix_tokens += timing["true_suffix_tokens"] state_times.append(time.perf_counter() - mark) elapsed = time.perf_counter() - started predictions[mode] = values - report["results"].append( - { - "mode": mode, - "wall_seconds": elapsed, - "decisions_per_second": len(values) / elapsed, - "state_p50_seconds": statistics.median(state_times), - "peak_cuda_bytes": torch.cuda.max_memory_allocated(), - } - ) - reference = {row["id"]: row for row in predictions["fresh"]} - report["comparisons_to_fresh"] = {} - for mode in ("serial_prefix", "parallel_shared"): - flips, maximum = [], 0.0 + record = { + "mode": mode, + "judgments": len(values), + "states": len(groups), + "wall_seconds": elapsed, + "judgments_per_second": len(values) / elapsed, + "state_latency_p50_seconds": statistics.median(state_times), + "state_latency_p95_seconds": percentile(state_times, 0.95), + "peak_cuda_bytes": torch.cuda.max_memory_allocated(), + } + if mode == "serial_prefix": + record["cache_hits"] = cache_hits + elif mode == "parallel_suffix": + record["padded_suffix_tokens"] = padded_suffix_tokens + record["true_suffix_tokens"] = true_suffix_tokens + report["results"].append(record) + reference = {row["id"]: row for row in predictions[MODES[0]]} + comparisons = {} + for mode in MODES[1:]: + flips, maximum, total, cells = [], 0.0, 0.0, 0 for row in predictions[mode]: old = reference[row["id"]] - maximum = max( - maximum, *(abs(a - b) for a, b in zip(old["probabilities"], row["probabilities"])) - ) + differences = [abs(a - b) for a, b in zip(old["probabilities"], row["probabilities"])] + maximum = max(maximum, *differences) + total += sum(differences) + cells += len(differences) if max(range(len(old["probabilities"])), key=old["probabilities"].__getitem__) != max( range(len(row["probabilities"])), key=row["probabilities"].__getitem__ ): flips.append(row["id"]) - report["comparisons_to_fresh"][mode] = { + comparisons[mode] = { + # Non-gating by design, so a pass can only mean an exact reproduction. + "passed": not flips and maximum == 0.0, "max_probability_difference": maximum, + "mean_probability_difference": total / cells, "argmax_flips": flips, + "tolerance": COMPARISON_TOLERANCE, } + comparisons["note"] = COMPARISON_NOTE + report["comparisons_to_fresh"] = comparisons args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text(json.dumps(report, indent=2, allow_nan=False) + "\n") args.output.with_suffix(".predictions.jsonl").write_text( "".join( - json.dumps({"mode": mode, **row}, allow_nan=False) + "\n" - for mode, values in predictions.items() - for row in values + json.dumps({"mode": PREDICTION_LABELS[mode], **row}, allow_nan=False) + "\n" + for mode in MODES + for row in predictions[mode] ) ) print(json.dumps(report["results"])) diff --git a/benchmarks/shape777_reranker.py b/benchmarks/shape777_reranker.py index 086a8b8..9830bb0 100644 --- a/benchmarks/shape777_reranker.py +++ b/benchmarks/shape777_reranker.py @@ -13,6 +13,36 @@ from semif_phase1.core import load_causal_model, softmax from semif_phase1.reranker import score_pair_batch +REPORT_VERSION = "shape777-reranker-v1" +TIMING_SCOPE = ( + "Warm loaded model; wall time includes prompt construction, tokenization, " + "padding, H2D, native forward, D2H and option normalization; excludes model " + "load and output write." +) +TOP_LEVEL_FIELDS = ( + "version", + "input_sha256", + "model", + "hardware", + "max_tokens", + "timing_scope", + "semantic_contract", + "results", +) +RESULT_FIELDS = ( + "pair_batch_size", + "judgments", + "states", + "wall_seconds", + "forward_seconds", + "judgments_per_second", + "state_latency_p50_seconds", + "state_latency_p95_seconds", + "padded_tokens", + "yes_no_pair_forwards", + "peak_cuda_bytes", +) + def percentile(values, fraction): ordered = sorted(values) @@ -49,10 +79,12 @@ def main() -> None: warm = next(iter(groups.values()))[0] score_pair_batch(model, tokenizer, [(warm, option) for option in warm["options"]], args.max_tokens) report = { - "version": "shape777-reranker-published-v1", + "version": REPORT_VERSION, "input_sha256": hashlib.sha256(args.input.read_bytes()).hexdigest(), "model": metadata, "hardware": torch.cuda.get_device_name(0), + "max_tokens": args.max_tokens, + "timing_scope": TIMING_SCOPE, "semantic_contract": ( "Two independent yes/no relevance passes per binary decision; " "option log-odds normalized only for relative comparison." @@ -94,12 +126,14 @@ def main() -> None: record = { "pair_batch_size": size, "judgments": len(predictions), + "states": len(groups), "wall_seconds": elapsed, "forward_seconds": forward_seconds, "judgments_per_second": len(predictions) / elapsed, "state_latency_p50_seconds": statistics.median(state_times), "state_latency_p95_seconds": percentile(state_times, 0.95), "padded_tokens": padded_tokens, + "yes_no_pair_forwards": sum(len(row["options"]) for row in rows), "peak_cuda_bytes": torch.cuda.max_memory_allocated(), } report["results"].append(record) diff --git a/tests/test_benchmark_schema.py b/tests/test_benchmark_schema.py new file mode 100644 index 0000000..2e7140d --- /dev/null +++ b/tests/test_benchmark_schema.py @@ -0,0 +1,80 @@ +"""Benchmark report schemas must stay consumable by verify_published.py.""" +import importlib.util +import json +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parent.parent +RAW = ROOT / "results" / "raw" + + +def _load(name): + spec = importlib.util.spec_from_file_location(name, ROOT / "benchmarks" / f"{name}.py") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +SHAPE = _load("shape777") +RERANKER = _load("shape777_reranker") + + +@pytest.fixture(scope="module") +def direct(): + return json.loads((RAW / "shape777-direct.json").read_text(encoding="utf-8")) + + +@pytest.fixture(scope="module") +def reranker(): + return json.loads((RAW / "shape777-reranker.json").read_text(encoding="utf-8")) + + +def test_direct_report_top_level_matches_the_script(direct): + assert list(direct) == list(SHAPE.TOP_LEVEL_FIELDS) + assert direct["version"] == SHAPE.REPORT_VERSION + assert direct["max_tokens"] == 4096 + assert direct["timing_scope"] == SHAPE.TIMING_SCOPE + + +def test_direct_report_modes_and_fields_match_the_script(direct): + assert [row["mode"] for row in direct["results"]] == list(SHAPE.MODES) + for row in direct["results"]: + assert set(row) == {"mode", *SHAPE.RESULT_FIELDS[row["mode"]]} + + +def test_direct_report_comparisons_match_the_script(direct): + comparisons = direct["comparisons_to_fresh"] + assert set(comparisons) == {*SHAPE.MODES[1:], "note"} + assert comparisons["note"] == SHAPE.COMPARISON_NOTE + for mode, record in comparisons.items(): + if mode == "note": + continue + assert list(record) == list(SHAPE.COMPARISON_FIELDS) + assert record["tolerance"] == SHAPE.COMPARISON_TOLERANCE + + +def test_reranker_report_top_level_matches_the_script(reranker): + assert list(reranker) == list(RERANKER.TOP_LEVEL_FIELDS) + assert reranker["version"] == RERANKER.REPORT_VERSION + assert reranker["max_tokens"] == 4096 + assert reranker["timing_scope"] == RERANKER.TIMING_SCOPE + + +def test_reranker_report_fields_match_the_script(reranker): + for row in reranker["results"]: + assert set(row) == set(RERANKER.RESULT_FIELDS) + + +def test_verifier_reads_only_keys_the_scripts_still_emit(): + verifier = (ROOT / "benchmarks" / "verify_published.py").read_text(encoding="utf-8") + for summary_key, raw_key in ( + ("wall_seconds", "wall_seconds"), + ("judgments_per_second", "judgments_per_second"), + ("state_p50_seconds", "state_latency_p50_seconds"), + ("peak_cuda_bytes", "peak_cuda_bytes"), + ): + assert f'("{summary_key}", "{raw_key}")' in verifier + for raw_key in ("judgments_per_second", "state_latency_p50_seconds"): + assert raw_key in SHAPE.RESULT_FIELDS["fresh_batch1"] + assert raw_key in RERANKER.RESULT_FIELDS