diff --git a/benchmarks/shape777.py b/benchmarks/shape777.py index feeb301..44c7867 100644 --- a/benchmarks/shape777.py +++ b/benchmarks/shape777.py @@ -15,6 +15,65 @@ from semif_phase1.serial import SerialPrefixScorer from semif_phase1.shared import score_shared +REPORT_VERSION = "shape777-direct-v1" +TIMING_SCOPE = ( + "Warm loaded model; complete mode wall time includes tokenization, H2D, forward " + "and D2H; excludes model load and output write." +) +COMPARISON_TOLERANCE = 1.0 +COMPARISON_NOTE = ( + "Tolerance is deliberately non-gating; report exact probability drift and all " + "argmax flips." +) +COMMON_FIELDS = ( + "judgments", + "states", + "wall_seconds", + "judgments_per_second", + "state_latency_p50_seconds", + "state_latency_p95_seconds", + "peak_cuda_bytes", +) +RESULT_FIELDS = { + "fresh_batch1": COMMON_FIELDS, + "serial_prefix": (*COMMON_FIELDS, "cache_hits"), + "parallel_suffix": (*COMMON_FIELDS, "padded_suffix_tokens", "true_suffix_tokens"), +} +COMPARISON_FIELDS = ( + "passed", + "max_probability_difference", + "mean_probability_difference", + "argmax_flips", + "tolerance", +) +# The report labels the unbatched mode as fresh_batch1 while its prediction rows +# were committed as fresh, so the two vocabularies are kept distinct. +MODES = ("fresh_batch1", "serial_prefix", "parallel_suffix") +PREDICTION_LABELS = { + "fresh_batch1": "fresh", + "serial_prefix": "serial_prefix", + "parallel_suffix": "parallel_suffix", +} +TOP_LEVEL_FIELDS = ( + "version", + "input_sha256", + "model", + "hardware", + "max_tokens", + "timing_scope", + "results", + "comparisons_to_fresh", +) + + +def percentile(values, fraction): + ordered = sorted(values) + index = (len(ordered) - 1) * fraction + lower = int(index) + return ordered[lower] + ( + ordered[min(lower + 1, len(ordered) - 1)] - ordered[lower] + ) * (index - lower) + def main() -> None: parser = argparse.ArgumentParser(description=__doc__) @@ -42,64 +101,87 @@ def main() -> None: warm_serial.score(row) score_shared(model, tokenizer, first, metadata, args.max_tokens) report = { - "version": "shape777-published-v1", + "version": REPORT_VERSION, "input_sha256": hashlib.sha256(args.input.read_bytes()).hexdigest(), "model": metadata, "hardware": torch.cuda.get_device_name(0), - "timing_scope": "Warm model; includes prompt construction, tokenization, transfers, forward passes and CPU readout.", + "max_tokens": args.max_tokens, + "timing_scope": TIMING_SCOPE, "results": [], } predictions = {} - for mode in ("fresh", "serial_prefix", "parallel_shared"): + for mode in MODES: torch.cuda.reset_peak_memory_stats() started = time.perf_counter() values, state_times = [], [] + cache_hits = 0 + padded_suffix_tokens = 0 + true_suffix_tokens = 0 for group in groups.values(): mark = time.perf_counter() - if mode == "fresh": + if mode == "fresh_batch1": values.extend(score(model, tokenizer, row, metadata, args.max_tokens) for row in group) elif mode == "serial_prefix": scorer = SerialPrefixScorer(model, tokenizer, metadata, args.max_tokens) - values.extend(scorer.score(row) for row in group) + for row in group: + result = scorer.score(row) + cache_hits += bool(result["cache_hit"]) + values.append(result) else: - scored, _ = score_shared(model, tokenizer, group, metadata, args.max_tokens) + scored, timing = score_shared(model, tokenizer, group, metadata, args.max_tokens) values.extend(scored) + padded_suffix_tokens += timing["padded_suffix_tokens"] + true_suffix_tokens += timing["true_suffix_tokens"] state_times.append(time.perf_counter() - mark) elapsed = time.perf_counter() - started predictions[mode] = values - report["results"].append( - { - "mode": mode, - "wall_seconds": elapsed, - "decisions_per_second": len(values) / elapsed, - "state_p50_seconds": statistics.median(state_times), - "peak_cuda_bytes": torch.cuda.max_memory_allocated(), - } - ) - reference = {row["id"]: row for row in predictions["fresh"]} - report["comparisons_to_fresh"] = {} - for mode in ("serial_prefix", "parallel_shared"): - flips, maximum = [], 0.0 + record = { + "mode": mode, + "judgments": len(values), + "states": len(groups), + "wall_seconds": elapsed, + "judgments_per_second": len(values) / elapsed, + "state_latency_p50_seconds": statistics.median(state_times), + "state_latency_p95_seconds": percentile(state_times, 0.95), + "peak_cuda_bytes": torch.cuda.max_memory_allocated(), + } + if mode == "serial_prefix": + record["cache_hits"] = cache_hits + elif mode == "parallel_suffix": + record["padded_suffix_tokens"] = padded_suffix_tokens + record["true_suffix_tokens"] = true_suffix_tokens + report["results"].append(record) + reference = {row["id"]: row for row in predictions[MODES[0]]} + comparisons = {} + for mode in MODES[1:]: + flips, maximum, total, cells = [], 0.0, 0.0, 0 for row in predictions[mode]: old = reference[row["id"]] - maximum = max( - maximum, *(abs(a - b) for a, b in zip(old["probabilities"], row["probabilities"])) - ) + differences = [abs(a - b) for a, b in zip(old["probabilities"], row["probabilities"])] + maximum = max(maximum, *differences) + total += sum(differences) + cells += len(differences) if max(range(len(old["probabilities"])), key=old["probabilities"].__getitem__) != max( range(len(row["probabilities"])), key=row["probabilities"].__getitem__ ): flips.append(row["id"]) - report["comparisons_to_fresh"][mode] = { + comparisons[mode] = { + # Non-gating by design, so a pass can only mean an exact reproduction. + "passed": not flips and maximum == 0.0, "max_probability_difference": maximum, + "mean_probability_difference": total / cells, "argmax_flips": flips, + "tolerance": COMPARISON_TOLERANCE, } + comparisons["note"] = COMPARISON_NOTE + report["comparisons_to_fresh"] = comparisons args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text(json.dumps(report, indent=2, allow_nan=False) + "\n") args.output.with_suffix(".predictions.jsonl").write_text( "".join( - json.dumps({"mode": mode, **row}, allow_nan=False) + "\n" - for mode, values in predictions.items() - for row in values + json.dumps({"mode": PREDICTION_LABELS[mode], **row}, allow_nan=False) + "\n" + for mode in MODES + for row in predictions[mode] ) ) print(json.dumps(report["results"])) diff --git a/benchmarks/shape777_reranker.py b/benchmarks/shape777_reranker.py index 086a8b8..9830bb0 100644 --- a/benchmarks/shape777_reranker.py +++ b/benchmarks/shape777_reranker.py @@ -13,6 +13,36 @@ from semif_phase1.core import load_causal_model, softmax from semif_phase1.reranker import score_pair_batch +REPORT_VERSION = "shape777-reranker-v1" +TIMING_SCOPE = ( + "Warm loaded model; wall time includes prompt construction, tokenization, " + "padding, H2D, native forward, D2H and option normalization; excludes model " + "load and output write." +) +TOP_LEVEL_FIELDS = ( + "version", + "input_sha256", + "model", + "hardware", + "max_tokens", + "timing_scope", + "semantic_contract", + "results", +) +RESULT_FIELDS = ( + "pair_batch_size", + "judgments", + "states", + "wall_seconds", + "forward_seconds", + "judgments_per_second", + "state_latency_p50_seconds", + "state_latency_p95_seconds", + "padded_tokens", + "yes_no_pair_forwards", + "peak_cuda_bytes", +) + def percentile(values, fraction): ordered = sorted(values) @@ -49,10 +79,12 @@ def main() -> None: warm = next(iter(groups.values()))[0] score_pair_batch(model, tokenizer, [(warm, option) for option in warm["options"]], args.max_tokens) report = { - "version": "shape777-reranker-published-v1", + "version": REPORT_VERSION, "input_sha256": hashlib.sha256(args.input.read_bytes()).hexdigest(), "model": metadata, "hardware": torch.cuda.get_device_name(0), + "max_tokens": args.max_tokens, + "timing_scope": TIMING_SCOPE, "semantic_contract": ( "Two independent yes/no relevance passes per binary decision; " "option log-odds normalized only for relative comparison." @@ -94,12 +126,14 @@ def main() -> None: record = { "pair_batch_size": size, "judgments": len(predictions), + "states": len(groups), "wall_seconds": elapsed, "forward_seconds": forward_seconds, "judgments_per_second": len(predictions) / elapsed, "state_latency_p50_seconds": statistics.median(state_times), "state_latency_p95_seconds": percentile(state_times, 0.95), "padded_tokens": padded_tokens, + "yes_no_pair_forwards": sum(len(row["options"]) for row in rows), "peak_cuda_bytes": torch.cuda.max_memory_allocated(), } report["results"].append(record) diff --git a/tests/test_benchmark_schema.py b/tests/test_benchmark_schema.py new file mode 100644 index 0000000..2e7140d --- /dev/null +++ b/tests/test_benchmark_schema.py @@ -0,0 +1,80 @@ +"""Benchmark report schemas must stay consumable by verify_published.py.""" +import importlib.util +import json +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parent.parent +RAW = ROOT / "results" / "raw" + + +def _load(name): + spec = importlib.util.spec_from_file_location(name, ROOT / "benchmarks" / f"{name}.py") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +SHAPE = _load("shape777") +RERANKER = _load("shape777_reranker") + + +@pytest.fixture(scope="module") +def direct(): + return json.loads((RAW / "shape777-direct.json").read_text(encoding="utf-8")) + + +@pytest.fixture(scope="module") +def reranker(): + return json.loads((RAW / "shape777-reranker.json").read_text(encoding="utf-8")) + + +def test_direct_report_top_level_matches_the_script(direct): + assert list(direct) == list(SHAPE.TOP_LEVEL_FIELDS) + assert direct["version"] == SHAPE.REPORT_VERSION + assert direct["max_tokens"] == 4096 + assert direct["timing_scope"] == SHAPE.TIMING_SCOPE + + +def test_direct_report_modes_and_fields_match_the_script(direct): + assert [row["mode"] for row in direct["results"]] == list(SHAPE.MODES) + for row in direct["results"]: + assert set(row) == {"mode", *SHAPE.RESULT_FIELDS[row["mode"]]} + + +def test_direct_report_comparisons_match_the_script(direct): + comparisons = direct["comparisons_to_fresh"] + assert set(comparisons) == {*SHAPE.MODES[1:], "note"} + assert comparisons["note"] == SHAPE.COMPARISON_NOTE + for mode, record in comparisons.items(): + if mode == "note": + continue + assert list(record) == list(SHAPE.COMPARISON_FIELDS) + assert record["tolerance"] == SHAPE.COMPARISON_TOLERANCE + + +def test_reranker_report_top_level_matches_the_script(reranker): + assert list(reranker) == list(RERANKER.TOP_LEVEL_FIELDS) + assert reranker["version"] == RERANKER.REPORT_VERSION + assert reranker["max_tokens"] == 4096 + assert reranker["timing_scope"] == RERANKER.TIMING_SCOPE + + +def test_reranker_report_fields_match_the_script(reranker): + for row in reranker["results"]: + assert set(row) == set(RERANKER.RESULT_FIELDS) + + +def test_verifier_reads_only_keys_the_scripts_still_emit(): + verifier = (ROOT / "benchmarks" / "verify_published.py").read_text(encoding="utf-8") + for summary_key, raw_key in ( + ("wall_seconds", "wall_seconds"), + ("judgments_per_second", "judgments_per_second"), + ("state_p50_seconds", "state_latency_p50_seconds"), + ("peak_cuda_bytes", "peak_cuda_bytes"), + ): + assert f'("{summary_key}", "{raw_key}")' in verifier + for raw_key in ("judgments_per_second", "state_latency_p50_seconds"): + assert raw_key in SHAPE.RESULT_FIELDS["fresh_batch1"] + assert raw_key in RERANKER.RESULT_FIELDS