Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
134 changes: 108 additions & 26 deletions benchmarks/shape777.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,65 @@
from semif_phase1.serial import SerialPrefixScorer
from semif_phase1.shared import score_shared

REPORT_VERSION = "shape777-direct-v1"
TIMING_SCOPE = (
"Warm loaded model; complete mode wall time includes tokenization, H2D, forward "
"and D2H; excludes model load and output write."
)
COMPARISON_TOLERANCE = 1.0
COMPARISON_NOTE = (
"Tolerance is deliberately non-gating; report exact probability drift and all "
"argmax flips."
)
COMMON_FIELDS = (
"judgments",
"states",
"wall_seconds",
"judgments_per_second",
"state_latency_p50_seconds",
"state_latency_p95_seconds",
"peak_cuda_bytes",
)
RESULT_FIELDS = {
"fresh_batch1": COMMON_FIELDS,
"serial_prefix": (*COMMON_FIELDS, "cache_hits"),
"parallel_suffix": (*COMMON_FIELDS, "padded_suffix_tokens", "true_suffix_tokens"),
}
COMPARISON_FIELDS = (
"passed",
"max_probability_difference",
"mean_probability_difference",
"argmax_flips",
"tolerance",
)
# The report labels the unbatched mode as fresh_batch1 while its prediction rows
# were committed as fresh, so the two vocabularies are kept distinct.
MODES = ("fresh_batch1", "serial_prefix", "parallel_suffix")
PREDICTION_LABELS = {
"fresh_batch1": "fresh",
"serial_prefix": "serial_prefix",
"parallel_suffix": "parallel_suffix",
}
TOP_LEVEL_FIELDS = (
"version",
"input_sha256",
"model",
"hardware",
"max_tokens",
"timing_scope",
"results",
"comparisons_to_fresh",
)


def percentile(values, fraction):
ordered = sorted(values)
index = (len(ordered) - 1) * fraction
lower = int(index)
return ordered[lower] + (
ordered[min(lower + 1, len(ordered) - 1)] - ordered[lower]
) * (index - lower)


def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
Expand Down Expand Up @@ -42,64 +101,87 @@ def main() -> None:
warm_serial.score(row)
score_shared(model, tokenizer, first, metadata, args.max_tokens)
report = {
"version": "shape777-published-v1",
"version": REPORT_VERSION,
"input_sha256": hashlib.sha256(args.input.read_bytes()).hexdigest(),
"model": metadata,
"hardware": torch.cuda.get_device_name(0),
"timing_scope": "Warm model; includes prompt construction, tokenization, transfers, forward passes and CPU readout.",
"max_tokens": args.max_tokens,
"timing_scope": TIMING_SCOPE,
"results": [],
}
predictions = {}
for mode in ("fresh", "serial_prefix", "parallel_shared"):
for mode in MODES:
torch.cuda.reset_peak_memory_stats()
started = time.perf_counter()
values, state_times = [], []
cache_hits = 0
padded_suffix_tokens = 0
true_suffix_tokens = 0
for group in groups.values():
mark = time.perf_counter()
if mode == "fresh":
if mode == "fresh_batch1":
values.extend(score(model, tokenizer, row, metadata, args.max_tokens) for row in group)
elif mode == "serial_prefix":
scorer = SerialPrefixScorer(model, tokenizer, metadata, args.max_tokens)
values.extend(scorer.score(row) for row in group)
for row in group:
result = scorer.score(row)
cache_hits += bool(result["cache_hit"])
values.append(result)
else:
scored, _ = score_shared(model, tokenizer, group, metadata, args.max_tokens)
scored, timing = score_shared(model, tokenizer, group, metadata, args.max_tokens)
values.extend(scored)
padded_suffix_tokens += timing["padded_suffix_tokens"]
true_suffix_tokens += timing["true_suffix_tokens"]
state_times.append(time.perf_counter() - mark)
elapsed = time.perf_counter() - started
predictions[mode] = values
report["results"].append(
{
"mode": mode,
"wall_seconds": elapsed,
"decisions_per_second": len(values) / elapsed,
"state_p50_seconds": statistics.median(state_times),
"peak_cuda_bytes": torch.cuda.max_memory_allocated(),
}
)
reference = {row["id"]: row for row in predictions["fresh"]}
report["comparisons_to_fresh"] = {}
for mode in ("serial_prefix", "parallel_shared"):
flips, maximum = [], 0.0
record = {
"mode": mode,
"judgments": len(values),
"states": len(groups),
"wall_seconds": elapsed,
"judgments_per_second": len(values) / elapsed,
"state_latency_p50_seconds": statistics.median(state_times),
"state_latency_p95_seconds": percentile(state_times, 0.95),
"peak_cuda_bytes": torch.cuda.max_memory_allocated(),
}
if mode == "serial_prefix":
record["cache_hits"] = cache_hits
elif mode == "parallel_suffix":
record["padded_suffix_tokens"] = padded_suffix_tokens
record["true_suffix_tokens"] = true_suffix_tokens
report["results"].append(record)
reference = {row["id"]: row for row in predictions[MODES[0]]}
comparisons = {}
for mode in MODES[1:]:
flips, maximum, total, cells = [], 0.0, 0.0, 0
for row in predictions[mode]:
old = reference[row["id"]]
maximum = max(
maximum, *(abs(a - b) for a, b in zip(old["probabilities"], row["probabilities"]))
)
differences = [abs(a - b) for a, b in zip(old["probabilities"], row["probabilities"])]
maximum = max(maximum, *differences)
total += sum(differences)
cells += len(differences)
if max(range(len(old["probabilities"])), key=old["probabilities"].__getitem__) != max(
range(len(row["probabilities"])), key=row["probabilities"].__getitem__
):
flips.append(row["id"])
report["comparisons_to_fresh"][mode] = {
comparisons[mode] = {
# Non-gating by design, so a pass can only mean an exact reproduction.
"passed": not flips and maximum == 0.0,
"max_probability_difference": maximum,
"mean_probability_difference": total / cells,
"argmax_flips": flips,
"tolerance": COMPARISON_TOLERANCE,
}
comparisons["note"] = COMPARISON_NOTE
report["comparisons_to_fresh"] = comparisons
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(report, indent=2, allow_nan=False) + "\n")
args.output.with_suffix(".predictions.jsonl").write_text(
"".join(
json.dumps({"mode": mode, **row}, allow_nan=False) + "\n"
for mode, values in predictions.items()
for row in values
json.dumps({"mode": PREDICTION_LABELS[mode], **row}, allow_nan=False) + "\n"
for mode in MODES
for row in predictions[mode]
)
)
print(json.dumps(report["results"]))
Expand Down
36 changes: 35 additions & 1 deletion benchmarks/shape777_reranker.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,36 @@
from semif_phase1.core import load_causal_model, softmax
from semif_phase1.reranker import score_pair_batch

REPORT_VERSION = "shape777-reranker-v1"
TIMING_SCOPE = (
"Warm loaded model; wall time includes prompt construction, tokenization, "
"padding, H2D, native forward, D2H and option normalization; excludes model "
"load and output write."
)
TOP_LEVEL_FIELDS = (
"version",
"input_sha256",
"model",
"hardware",
"max_tokens",
"timing_scope",
"semantic_contract",
"results",
)
RESULT_FIELDS = (
"pair_batch_size",
"judgments",
"states",
"wall_seconds",
"forward_seconds",
"judgments_per_second",
"state_latency_p50_seconds",
"state_latency_p95_seconds",
"padded_tokens",
"yes_no_pair_forwards",
"peak_cuda_bytes",
)


def percentile(values, fraction):
ordered = sorted(values)
Expand Down Expand Up @@ -49,10 +79,12 @@ def main() -> None:
warm = next(iter(groups.values()))[0]
score_pair_batch(model, tokenizer, [(warm, option) for option in warm["options"]], args.max_tokens)
report = {
"version": "shape777-reranker-published-v1",
"version": REPORT_VERSION,
"input_sha256": hashlib.sha256(args.input.read_bytes()).hexdigest(),
"model": metadata,
"hardware": torch.cuda.get_device_name(0),
"max_tokens": args.max_tokens,
"timing_scope": TIMING_SCOPE,
"semantic_contract": (
"Two independent yes/no relevance passes per binary decision; "
"option log-odds normalized only for relative comparison."
Expand Down Expand Up @@ -94,12 +126,14 @@ def main() -> None:
record = {
"pair_batch_size": size,
"judgments": len(predictions),
"states": len(groups),
"wall_seconds": elapsed,
"forward_seconds": forward_seconds,
"judgments_per_second": len(predictions) / elapsed,
"state_latency_p50_seconds": statistics.median(state_times),
"state_latency_p95_seconds": percentile(state_times, 0.95),
"padded_tokens": padded_tokens,
"yes_no_pair_forwards": sum(len(row["options"]) for row in rows),
"peak_cuda_bytes": torch.cuda.max_memory_allocated(),
}
report["results"].append(record)
Expand Down
80 changes: 80 additions & 0 deletions tests/test_benchmark_schema.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,80 @@
"""Benchmark report schemas must stay consumable by verify_published.py."""
import importlib.util
import json
from pathlib import Path

import pytest

ROOT = Path(__file__).resolve().parent.parent
RAW = ROOT / "results" / "raw"


def _load(name):
spec = importlib.util.spec_from_file_location(name, ROOT / "benchmarks" / f"{name}.py")
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module


SHAPE = _load("shape777")
RERANKER = _load("shape777_reranker")


@pytest.fixture(scope="module")
def direct():
return json.loads((RAW / "shape777-direct.json").read_text(encoding="utf-8"))


@pytest.fixture(scope="module")
def reranker():
return json.loads((RAW / "shape777-reranker.json").read_text(encoding="utf-8"))


def test_direct_report_top_level_matches_the_script(direct):
assert list(direct) == list(SHAPE.TOP_LEVEL_FIELDS)
assert direct["version"] == SHAPE.REPORT_VERSION
assert direct["max_tokens"] == 4096
assert direct["timing_scope"] == SHAPE.TIMING_SCOPE


def test_direct_report_modes_and_fields_match_the_script(direct):
assert [row["mode"] for row in direct["results"]] == list(SHAPE.MODES)
for row in direct["results"]:
assert set(row) == {"mode", *SHAPE.RESULT_FIELDS[row["mode"]]}


def test_direct_report_comparisons_match_the_script(direct):
comparisons = direct["comparisons_to_fresh"]
assert set(comparisons) == {*SHAPE.MODES[1:], "note"}
assert comparisons["note"] == SHAPE.COMPARISON_NOTE
for mode, record in comparisons.items():
if mode == "note":
continue
assert list(record) == list(SHAPE.COMPARISON_FIELDS)
assert record["tolerance"] == SHAPE.COMPARISON_TOLERANCE


def test_reranker_report_top_level_matches_the_script(reranker):
assert list(reranker) == list(RERANKER.TOP_LEVEL_FIELDS)
assert reranker["version"] == RERANKER.REPORT_VERSION
assert reranker["max_tokens"] == 4096
assert reranker["timing_scope"] == RERANKER.TIMING_SCOPE


def test_reranker_report_fields_match_the_script(reranker):
for row in reranker["results"]:
assert set(row) == set(RERANKER.RESULT_FIELDS)


def test_verifier_reads_only_keys_the_scripts_still_emit():
verifier = (ROOT / "benchmarks" / "verify_published.py").read_text(encoding="utf-8")
for summary_key, raw_key in (
("wall_seconds", "wall_seconds"),
("judgments_per_second", "judgments_per_second"),
("state_p50_seconds", "state_latency_p50_seconds"),
("peak_cuda_bytes", "peak_cuda_bytes"),
):
assert f'("{summary_key}", "{raw_key}")' in verifier
for raw_key in ("judgments_per_second", "state_latency_p50_seconds"):
assert raw_key in SHAPE.RESULT_FIELDS["fresh_batch1"]
assert raw_key in RERANKER.RESULT_FIELDS