Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions benchmarks/conformance/.gitignore
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
.venv/
__pycache__/
.pytest_cache/
*.egg-info/
private/
*.pt
*.so
51 changes: 51 additions & 0 deletions benchmarks/conformance/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
# Numerical and state conformance support

This is the portable support used in the
[M1/M8 and eager/compiled investigation](https://github.com/Terrydaktal/d7-rdna4-report/blob/main/reports/d7-rdna4-2026-09-17/REPORT.md).
It compares implementations against a declared reference; passing finite samples
does not prove arbitrary-input equivalence or model task accuracy.

```text
conformance/
├── src/qwen_r9700_lab/ # preserved module names and source identities
├── tests/ # synthetic CPU regressions and negative controls
├── configs/profiles/ # explicit serial GDN arithmetic contract
├── SOURCE_PROVENANCE.json
├── pyproject.toml
└── uv.lock
```

Install and run the CPU suite without loading a model:

```sh
uv sync --project benchmarks/conformance --group dev
uv run --project benchmarks/conformance pytest benchmarks/conformance/tests -q
```

The top-k comparator consumes aligned full-logit summaries and reports top-1,
top-10 and top-20 set agreement, order agreement, overlap, retained-score equality,
boundary ties and full-row digests separately. Empty domains, mismatched positions,
changed manifests and incomplete evidence are errors. It does not decode tokens.

The other modules provide logical-state comparisons, transition invariants,
tentative-output checking, fault injection, source/binary binding, exclusive GPU
leases and checkpointed worker supervision. Abstract proofs and CPU simulations
remain distinct from native GPU qualification. Private raw rows and public
aggregate receipts have different storage rules.

For native replay, the companion `benchmarks/d7-repair` submission supplies the
actual source-bound workers and numerical repair adapters. Build and qualification
receipts must match before those adapters can be installed. Production defaults
and container builds are unaffected by this support package.

The Python namespace is retained to preserve compatibility with existing sealed
workers. Renaming and narrowing this dependency closure can be reviewed separately
from the initial import. The standalone CPU suite is the publication-time check;
the report describes the separate completed pinned-stack GPU campaign.

The companion D7 repair bundle also supplies the source-preserving native call
tape and four-column isolated-stage study. Its evidence auditor rejects missing
layers, changed fixtures, unchecked reference output and failed state recovery.
The combined support/repair CPU suite passes **412 tests**, with one retained
compiler-artifact check skipped. Native qualification is reported separately in
the [public report](https://github.com/Terrydaktal/d7-rdna4-report/blob/main/reports/d7-rdna4-2026-09-17/REPORT.md).
58 changes: 58 additions & 0 deletions benchmarks/conformance/SOURCE_PROVENANCE.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,58 @@
{
"files": {
"src/qwen_r9700_lab/config.py": "799d3612a07a431463abaac657fcdcaabf749bd034d2767a212b98fef134c9ed",
"src/qwen_r9700_lab/conformance_artifacts.py": "4067a3648ab3a7cf08179d9848304f05f019c09a49654f6306a9020d48e0b704",
"src/qwen_r9700_lab/conformance_attention_cut.py": "74f810ca277fad47a7d8ca3877d5104368e01ca8493eaa216b6b9ab07694a0fb",
"src/qwen_r9700_lab/conformance_boundaries.py": "4548064e635243f945e443222809ef7f858b366cd545809bc2672c433c4c9c39",
"src/qwen_r9700_lab/conformance_campaign.py": "d8d8fbd45c9767f8bec07a96e65453e9e6c938289720be03587a60f8a011169d",
"src/qwen_r9700_lab/conformance_cli.py": "7228102934f5364dad27a639c8e3e860fbb07d2bc574db878e6849a15656d212",
"src/qwen_r9700_lab/conformance_control.py": "0ff173eeb849d6fa6f61960c3a08aee06508329cb4f1068a9ba7cc427da43ad3",
"src/qwen_r9700_lab/conformance_dispatch.py": "9e3ecdbf1a81e9123c7edabce425e442545725b312b5a2282573cbef14ee30f8",
"src/qwen_r9700_lab/conformance_execution_modes.py": "f1f651829bb1177e6acc6695396615ac17f3d91e59fe4f058af8cdd119c7817a",
"src/qwen_r9700_lab/conformance_faults.py": "22b348e8ca1577b5e66debbcde09890d6432f0230d56252d3e1b6dcb10a22366",
"src/qwen_r9700_lab/conformance_gate.py": "495e5f04a756c479131f15db11d9300a59ceda75cdfe332b688a668b69ffd998",
"src/qwen_r9700_lab/conformance_gdn_contract.py": "7c84ce706a87f7bb6264c3d57d1e005f9c07bf57f53bdba97ea5e1d102a406ed",
"src/qwen_r9700_lab/conformance_gpu_lease.py": "d224177846aab2d068de543ca35ae3473c752f0329a5490a9f9d9ef0a97c9622",
"src/qwen_r9700_lab/conformance_instrumentation.py": "e1f1c7d4ab4d7b73f535e177809ffe25546209352af6b67e76c21a35a55f96eb",
"src/qwen_r9700_lab/conformance_invariants.py": "ec146c9785835d244eefd8562fc23aff1f8899676351515f7921dbe517561394",
"src/qwen_r9700_lab/conformance_lifecycle.py": "5c38d401e2e01102d2194ad8e1427eb39b72759e3ea2346fdf9dd76d1e3f5727",
"src/qwen_r9700_lab/conformance_mode_boundaries.py": "b50dcf80a9a6c6de5cae6d30779d6feefe4edeed24b4825e20b74baadac24275",
"src/qwen_r9700_lab/conformance_model.py": "3e0d2cca336b3f86b9f3a552a716519a359995a99e28e5c9658b1732b3210f44",
"src/qwen_r9700_lab/conformance_native_reference.py": "e1e3b4cebe62e46a398cd6070656c45ea756335121941a6759711677195746a3",
"src/qwen_r9700_lab/conformance_obligations.py": "ba3f91e6c473302998eec3b087daeaef9578f6395286c0c91dc6cd1915af58df",
"src/qwen_r9700_lab/conformance_observer.py": "d3a73a96e0871c53d7a198e36925c90ef4095cdfe6feda69b14a8549b100ad26",
"src/qwen_r9700_lab/conformance_parser.py": "6964d502c8b3cc51fc910237f6b5a8d476cab8de9ac9b40b4dbb13942820048a",
"src/qwen_r9700_lab/conformance_precision_intervention.py": "1212c666c14a72982f948f7a145074150117fc8f2315e3289ad40171892bbafe",
"src/qwen_r9700_lab/conformance_priority.py": "1cf992191a1736ee4848252091a8dbba56bbdb6a9c35cd90e14317dab7ec3926",
"src/qwen_r9700_lab/conformance_proofs.py": "979a10bd8203f4c3142f36ef9a40bf1532c185e474d3adbf677fdb49cef05428",
"src/qwen_r9700_lab/conformance_protocol.py": "c290f71871d34dcce9fe277da682a849c6c24f2358944852cdc39a48711dc89f",
"src/qwen_r9700_lab/conformance_queue.py": "d1ccf1d9d73cf703caf218324ca5f03767910e25ef1e73e903f23884426cdf95",
"src/qwen_r9700_lab/conformance_radiance.py": "d4980810ce03b2aa451738f321ba4346e5b8192f3034605b08674ab3888568a3",
"src/qwen_r9700_lab/conformance_reference.py": "0bc2e74d1295d53d839c822661e42ce8bddd36e63e36853418e9544187aa1e04",
"src/qwen_r9700_lab/conformance_reference_store.py": "720258558c8152c2f8e99035926a3abee95c111359c1a323e7da9098ba3d10b1",
"src/qwen_r9700_lab/conformance_replay.py": "178fcf75d2e88979c34085fba110df1df25277bbd30af543c905e109ce3a3834",
"src/qwen_r9700_lab/conformance_rotary_intervention.py": "a4ec7ce8cfebb0d715c5c259f6eb092dcf8854aeeee42e01a7f54c5988a19c57",
"src/qwen_r9700_lab/conformance_rotary_repair.py": "98f464b0dd53c2bbb44a4cc9e9307d520202a81a9a3175a3c091c8d95da050a0",
"src/qwen_r9700_lab/conformance_runtime.py": "830e09e70d609c28abe3e23fa13b6cb96176c70cb5aeba0da419029338d967ef",
"src/qwen_r9700_lab/conformance_scenarios.py": "57d5f84bd1db875a13f3c61fe65f5118de388e2736bb0e865558ee624472351f",
"src/qwen_r9700_lab/conformance_session.py": "3069b56e1af18dbc5db365968807459e45936f67ae0d1f8731250b02450b81b6",
"src/qwen_r9700_lab/conformance_shm.py": "eab5f933b6ce3be6f0942713230699d31ecfb4a73101d5776a523aa774b47e8c",
"src/qwen_r9700_lab/conformance_stage_isolation.py": "312a7e8353530e2807b89ecc5111a7de5bce3cc2dd56f850dd662490fd860149",
"src/qwen_r9700_lab/conformance_state.py": "42ade68e94bf53594b53b17ee7e916a77c443c18fe4b90a795e7ce510ac38c69",
"src/qwen_r9700_lab/conformance_topk.py": "940edebf21f2e8dd335ad52d37d7507d86a41e7fafc8884a5499e704069f30ce",
"src/qwen_r9700_lab/conformance_transport.py": "2bcf4eba295ea245c68d4f8bb8af28b68e60397086df72435ebd11973079cdef",
"src/qwen_r9700_lab/diagnostic_contract.py": "603680020b6a69d728d615eef9815b424610b4979410f223ab76e28dc4ee779e",
"src/qwen_r9700_lab/exact_fp8_metrics.py": "6c7aacb3a50e0f17c3f4eafdcd90b64b5efab8a63c9e0e8e57be98431d94a2bd",
"src/qwen_r9700_lab/manifests.py": "1eaea2743bcd851d54e6fc4809b5a26e9ab33a755fdbf547264183fc70ec6485",
"src/qwen_r9700_lab/ordered_reference_linear.c": "314ff8a91cd06dd14eafd5ee5695a2d90f77c5186820797a4460cb182fb3d272",
"src/qwen_r9700_lab/ordered_reference_linear.py": "7a8d5559918a3a21f2f0491f26942455f14aa0fe71b8a6029d4fcb80c4319bc7",
"src/qwen_r9700_lab/radiance_cache.py": "04adc11af696ab006570ee481b9e0b04d673e4bc6282469ea03061e6567abe03",
"tests/test_conformance_execution_modes.py": "cf4d3d64365c7797460f02a152bb43f50875a9ef3f79a1a4bed1d4f8d02c98b6",
"tests/test_conformance_rotary_repair.py": "0d5b771d005e805f4f3f0251ed82aab6ad1c39c0182f0811c7614de3f69e312b",
"tests/test_conformance_stage_isolation.py": "08e448a1620959e153907ad4b9d8d119a65b50990430946aebabab34c9c9df63",
"tests/test_d7_precision_comparison.py": "b40d22a9f2f27c63d0933b2f7a7e26f5804a3f9a406a2e28c5743c2032972902",
"tests/test_d7_rotary_comparison.py": "c78dd032b9aa9173da36fe5e16841f454207a8ca806065e6225d36d8c3a62e03"
},
"origin": "https://github.com/Terrydaktal/qwen-r9700-lab",
"report": "https://github.com/Terrydaktal/d7-rdna4-report/blob/main/reports/d7-rdna4-2026-09-17/REPORT.md"
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,152 @@
{
"schema": "urn:qwen:gdn-arithmetic-contract:v1",
"id": "stock-packed-gdn-m1-gfx1201-v1",
"scope": "GDN core from raw convolution Q/K/V and gate projections to output plus recurrent state; other operators have separate contracts.",
"status": "DECLARED; native cross-mode equivalence UNPROVED",
"selection": "Pinned stock vLLM packed single-token GDN, applied causally to each materialized token. R4D and NumPy remain separately identified implementations.",
"domain": {
"query_heads": 16,
"value_heads": 48,
"key_width": 128,
"value_width": 128,
"batch_sequences": 1,
"qkv_width": 10240,
"state_order": [
"head",
"value",
"key"
],
"state_elements": 786432,
"inputs": "Finite BF16 Q/K/V, a, b, dt_bias; finite FP32 A_log and initial state. Invalid geometry, aliasing, nonfinite output/state or incomplete evidence fails closed."
},
"precision": {
"qkv": "BF16 widened exactly to FP32",
"a_b_dt_bias": "BF16 widened exactly to FP32",
"A_log": "FP32",
"qk_normalization": "FP32 throughout; no intermediate BF16 cast",
"beta": "sigmoid in the pinned FP32 kernel, then BF16 round-to-nearest ties-even, then exact FP32 widening",
"decay_gate": "FP32",
"recurrent_state": "FP32 on every read, update and committed write",
"output": "BF16 round-to-nearest ties-even"
},
"operations": [
{
"name": "normalize",
"formula": "n(x) = x / sqrt(sum(x_i*x_i) + float32(1e-6))",
"arithmetic": "Pinned compiled reduction, square root and division. No replacement with a differently rounded reciprocal-square-root multiply."
},
{
"name": "scale_query",
"formula": "q = float32(n(Q) * float32(128**-0.5)); k = n(K)",
"arithmetic": "Scale q before the output dot product; not after it."
},
{
"name": "gates",
"formula": "x = float32(a + dt_bias); softplus = x if x > 20 else log(1 + exp(x)); g = -exp(A_log)*softplus; beta = widen(BF16_RNE(sigmoid(b)))",
"arithmetic": "The pinned kernel intrinsics and their compiled implementations define rounding. No cumulative-gate reconstruction or future-dependent fallback."
},
{
"name": "decay",
"formula": "D = float32(exp(g)*S)"
},
{
"name": "prediction",
"formula": "p = dot_stock(D,k)"
},
{
"name": "residual",
"formula": "r = float32(float32(v-p)*beta)"
},
{
"name": "state",
"formula": "S_next = update_stock(D,r,k)",
"arithmetic": "Retain the compiled multiply-add contraction; separate multiply/add is a different numerical path."
},
{
"name": "output",
"formula": "o = BF16_RNE(dot_stock(S_next,q))"
}
],
"arithmetic_authority": {
"kind": "pinned-native-executable",
"note": "Real-number formulae describe dataflow. Exact finite-precision results are defined by the retained kernel artifact and packing below, under the declared compiler/hardware assumptions. Source inspection or matching hashes is not proof of runtime dispatch or functional correctness.",
"kernel": "fused_recurrent_gated_delta_rule_packed_decode_kernel",
"source": {
"fused_recurrent.py": "00a3b971b0dbb6ed26e246970a0e1a21a9a174030974685bdd3f0c8ab5fb4bfe",
"fla_op.py": "456da84fd12411cf53c96a90dd4f78f49afb454c3d51d2c524ea8e7106fe8ae5"
},
"cache_entry": "XW6RXQHLHKDBX2FF2LE7HICYZGQCSR5L4HJZ7USZF4L2FZQULJTQ",
"files": {
"fused_recurrent_gated_delta_rule_packed_decode_kernel.source": "f47615b4e3c0a4136cb3d5f7e526f222c8609e73dff2d89111c34ba1745ac1a7",
"fused_recurrent_gated_delta_rule_packed_decode_kernel.ttir": "0333349d4ff544078a5d35354a5ca42d7fe37360af7ce188d9778dba7887aabf",
"fused_recurrent_gated_delta_rule_packed_decode_kernel.ttgir": "889064969acdafd8931d3d7ae02e5b0ce98fdb80ac1f9df9fb919b5259c2645c",
"fused_recurrent_gated_delta_rule_packed_decode_kernel.llir": "eb6f3975ed77e9f0b53975d1d1c0739abdba06813e64bea8aa6f66baeadd19f2",
"fused_recurrent_gated_delta_rule_packed_decode_kernel.amdgcn": "f07d71ae362c0e87633253a85c2571e0ac626ff8b520034e16b554687ec4e1a1",
"fused_recurrent_gated_delta_rule_packed_decode_kernel.hsaco": "9feeb930b8635122c552ad727fbe897d0854332f9a7880468662b9ed83350135",
"fused_recurrent_gated_delta_rule_packed_decode_kernel.json": "de1084d3b9e844ded01cd70f866eda11618a6e95e1471b0ccc0f295921c96fad"
},
"target": {
"backend": "hip",
"arch": "gfx1201",
"warp_size": 32
},
"compiler": {
"triton_version": "3.7.1",
"num_warps": 1,
"num_stages": 3,
"enable_fp_fusion": true,
"allow_flush_denorm": false,
"sanitize_overflow": true
},
"environment": {
"FLA_USE_FAST_OPS": "0"
},
"packing": {
"query_heads": 16,
"value_heads": 48,
"K": 128,
"V": 128,
"BK": 128,
"BV": 32,
"state_token_stride": 802816,
"a_token_stride": 96,
"b_token_stride": 96,
"index_stride": 1,
"state_slot": 1,
"grid": [
4,
48
],
"scale_fp32_hex": "f304b53d",
"note": "Single row makes the packed-QKV row stride irrelevant to this invocation. Reject or separately qualify other physical packing; physical block numbers are not semantic state."
}
},
"execution_modes": {
"m1": "One invocation F(S,u) with exactly one materialized input token.",
"prefill": "Ordered fold of that same F over tokens. Empty segment is the identity. Every intermediate prefix must match.",
"d7": "Verification rows are u[0]=previously emitted pending token, u[1:8]=seven proposals. For accepted proposal count k in 0..7, commit the state after rows 0..k (k+1 processed rows); the newly emitted correction/bonus token stays pending.",
"rollback": "Rejected rows cannot affect committed GDN state, convolution history, KV, positions or events. The GDN-only contract does not certify those other components.",
"snapshot": "Persist the contract identity with state; do not treat states created under a different arithmetic contract as interchangeable without checking or rebuilding."
},
"obligations": {
"partition": "fold(F,S,A+B) == fold(F,fold(F,S,A).state,B) for outputs and state, including empty segments",
"causal_prefix": "Changing inputs after prefix p leaves every output/state through p byte-identical.",
"accepted_prefix": "D7_commit(S,u,k) == fold(F,S,u[:k+1]).state with a separate pending-token field.",
"exactness": "Compare BF16 output bits and every FP32 logical-state bit; hashes and error tolerances alone do not establish equality."
},
"separate_targets": {
"numpy-serial-fp32-v1": "Independent canonical model. It rounds normalized Q/K to BF16, uses ordered non-fused reductions and scales the final output dot. It is not relabelled as this stock kernel.",
"r4d-unfused-m1": "Retains FP32 beta, uses its own reduction/normalization intrinsics, disables FMA and rounds output ties away. Existing prefill candidate targets this path, not this contract.",
"huggingface-fallback": "The pinned Transformers fallback also returns BF16 beta, but its normalization and framework reductions differ. Full bit equality with it is UNPROVED."
},
"required_qualification": [
"Reproduce the preserved stock first transition exactly with the selected invocation packing.",
"Compare repaired prefill/M1/D7 against this oracle from equal independent initial state.",
"Exercise every accepted width, all-prefix equality, future-suffix mutations, padding, nonzero state, extreme gates and BF16 halfway boundaries.",
"Then qualify graph execution, full-model logits/state, snapshots and long contexts; benchmark separately."
],
"production_changed": false,
"existing_matrix_contract_changed": false,
"native_repair_qualified": false,
"sha256": "d8149b00210c94acfca4f1f66030de302dcaf8155894de6a23d3f4b76da61e00"
}
38 changes: 38 additions & 0 deletions benchmarks/conformance/pyproject.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,38 @@
[project]
name = "radiance-conformance"
version = "0.1.0"
requires-python = ">=3.12"
dependencies = [
"numpy>=2",
"jsonschema>=4.23,<5",
"PyYAML>=6",
"psutil>=6",
"requests>=2",
"safetensors>=0.5",
]

[project.optional-dependencies]
cpu-tests = [
"torch>=2.14.0",
]

[dependency-groups]
dev = ["pytest>=8", "z3-solver>=4.13"]

[build-system]
requires = ["hatchling"]
build-backend = "hatchling.build"

[tool.hatch.build.targets.wheel]
packages = ["src/qwen_r9700_lab"]

[[tool.uv.index]]
name = "pytorch-cpu"
url = "https://download.pytorch.org/whl/cpu"
explicit = true

[tool.pytest.ini_options]
pythonpath = ["src", "tests"]

[tool.uv.sources]
torch = { index = "pytorch-cpu" }
1 change: 1 addition & 0 deletions benchmarks/conformance/src/qwen_r9700_lab/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
"""Source-bound conformance helpers; importing them does not activate inference."""
Loading