diff --git a/benchmarks/conformance/.gitignore b/benchmarks/conformance/.gitignore new file mode 100644 index 0000000..a23b143 --- /dev/null +++ b/benchmarks/conformance/.gitignore @@ -0,0 +1,7 @@ +.venv/ +__pycache__/ +.pytest_cache/ +*.egg-info/ +private/ +*.pt +*.so diff --git a/benchmarks/conformance/README.md b/benchmarks/conformance/README.md new file mode 100644 index 0000000..75a2257 --- /dev/null +++ b/benchmarks/conformance/README.md @@ -0,0 +1,51 @@ +# Numerical and state conformance support + +This is the portable support used in the +[M1/M8 and eager/compiled investigation](https://github.com/Terrydaktal/d7-rdna4-report/blob/main/reports/d7-rdna4-2026-09-17/REPORT.md). +It compares implementations against a declared reference; passing finite samples +does not prove arbitrary-input equivalence or model task accuracy. + +```text +conformance/ +├── src/qwen_r9700_lab/ # preserved module names and source identities +├── tests/ # synthetic CPU regressions and negative controls +├── configs/profiles/ # explicit serial GDN arithmetic contract +├── SOURCE_PROVENANCE.json +├── pyproject.toml +└── uv.lock +``` + +Install and run the CPU suite without loading a model: + +```sh +uv sync --project benchmarks/conformance --group dev +uv run --project benchmarks/conformance pytest benchmarks/conformance/tests -q +``` + +The top-k comparator consumes aligned full-logit summaries and reports top-1, +top-10 and top-20 set agreement, order agreement, overlap, retained-score equality, +boundary ties and full-row digests separately. Empty domains, mismatched positions, +changed manifests and incomplete evidence are errors. It does not decode tokens. + +The other modules provide logical-state comparisons, transition invariants, +tentative-output checking, fault injection, source/binary binding, exclusive GPU +leases and checkpointed worker supervision. Abstract proofs and CPU simulations +remain distinct from native GPU qualification. Private raw rows and public +aggregate receipts have different storage rules. + +For native replay, the companion `benchmarks/d7-repair` submission supplies the +actual source-bound workers and numerical repair adapters. Build and qualification +receipts must match before those adapters can be installed. Production defaults +and container builds are unaffected by this support package. + +The Python namespace is retained to preserve compatibility with existing sealed +workers. Renaming and narrowing this dependency closure can be reviewed separately +from the initial import. The standalone CPU suite is the publication-time check; +the report describes the separate completed pinned-stack GPU campaign. + +The companion D7 repair bundle also supplies the source-preserving native call +tape and four-column isolated-stage study. Its evidence auditor rejects missing +layers, changed fixtures, unchecked reference output and failed state recovery. +The combined support/repair CPU suite passes **412 tests**, with one retained +compiler-artifact check skipped. Native qualification is reported separately in +the [public report](https://github.com/Terrydaktal/d7-rdna4-report/blob/main/reports/d7-rdna4-2026-09-17/REPORT.md). diff --git a/benchmarks/conformance/SOURCE_PROVENANCE.json b/benchmarks/conformance/SOURCE_PROVENANCE.json new file mode 100644 index 0000000..9eb9df5 --- /dev/null +++ b/benchmarks/conformance/SOURCE_PROVENANCE.json @@ -0,0 +1,58 @@ +{ + "files": { + "src/qwen_r9700_lab/config.py": "799d3612a07a431463abaac657fcdcaabf749bd034d2767a212b98fef134c9ed", + "src/qwen_r9700_lab/conformance_artifacts.py": "4067a3648ab3a7cf08179d9848304f05f019c09a49654f6306a9020d48e0b704", + "src/qwen_r9700_lab/conformance_attention_cut.py": "74f810ca277fad47a7d8ca3877d5104368e01ca8493eaa216b6b9ab07694a0fb", + "src/qwen_r9700_lab/conformance_boundaries.py": "4548064e635243f945e443222809ef7f858b366cd545809bc2672c433c4c9c39", + "src/qwen_r9700_lab/conformance_campaign.py": "d8d8fbd45c9767f8bec07a96e65453e9e6c938289720be03587a60f8a011169d", + "src/qwen_r9700_lab/conformance_cli.py": "7228102934f5364dad27a639c8e3e860fbb07d2bc574db878e6849a15656d212", + "src/qwen_r9700_lab/conformance_control.py": "0ff173eeb849d6fa6f61960c3a08aee06508329cb4f1068a9ba7cc427da43ad3", + "src/qwen_r9700_lab/conformance_dispatch.py": "9e3ecdbf1a81e9123c7edabce425e442545725b312b5a2282573cbef14ee30f8", + "src/qwen_r9700_lab/conformance_execution_modes.py": "f1f651829bb1177e6acc6695396615ac17f3d91e59fe4f058af8cdd119c7817a", + "src/qwen_r9700_lab/conformance_faults.py": "22b348e8ca1577b5e66debbcde09890d6432f0230d56252d3e1b6dcb10a22366", + "src/qwen_r9700_lab/conformance_gate.py": "495e5f04a756c479131f15db11d9300a59ceda75cdfe332b688a668b69ffd998", + "src/qwen_r9700_lab/conformance_gdn_contract.py": "7c84ce706a87f7bb6264c3d57d1e005f9c07bf57f53bdba97ea5e1d102a406ed", + "src/qwen_r9700_lab/conformance_gpu_lease.py": "d224177846aab2d068de543ca35ae3473c752f0329a5490a9f9d9ef0a97c9622", + "src/qwen_r9700_lab/conformance_instrumentation.py": "e1f1c7d4ab4d7b73f535e177809ffe25546209352af6b67e76c21a35a55f96eb", + "src/qwen_r9700_lab/conformance_invariants.py": "ec146c9785835d244eefd8562fc23aff1f8899676351515f7921dbe517561394", + "src/qwen_r9700_lab/conformance_lifecycle.py": "5c38d401e2e01102d2194ad8e1427eb39b72759e3ea2346fdf9dd76d1e3f5727", + "src/qwen_r9700_lab/conformance_mode_boundaries.py": "b50dcf80a9a6c6de5cae6d30779d6feefe4edeed24b4825e20b74baadac24275", + "src/qwen_r9700_lab/conformance_model.py": "3e0d2cca336b3f86b9f3a552a716519a359995a99e28e5c9658b1732b3210f44", + "src/qwen_r9700_lab/conformance_native_reference.py": "e1e3b4cebe62e46a398cd6070656c45ea756335121941a6759711677195746a3", + "src/qwen_r9700_lab/conformance_obligations.py": "ba3f91e6c473302998eec3b087daeaef9578f6395286c0c91dc6cd1915af58df", + "src/qwen_r9700_lab/conformance_observer.py": "d3a73a96e0871c53d7a198e36925c90ef4095cdfe6feda69b14a8549b100ad26", + "src/qwen_r9700_lab/conformance_parser.py": "6964d502c8b3cc51fc910237f6b5a8d476cab8de9ac9b40b4dbb13942820048a", + "src/qwen_r9700_lab/conformance_precision_intervention.py": "1212c666c14a72982f948f7a145074150117fc8f2315e3289ad40171892bbafe", + "src/qwen_r9700_lab/conformance_priority.py": "1cf992191a1736ee4848252091a8dbba56bbdb6a9c35cd90e14317dab7ec3926", + "src/qwen_r9700_lab/conformance_proofs.py": "979a10bd8203f4c3142f36ef9a40bf1532c185e474d3adbf677fdb49cef05428", + "src/qwen_r9700_lab/conformance_protocol.py": "c290f71871d34dcce9fe277da682a849c6c24f2358944852cdc39a48711dc89f", + "src/qwen_r9700_lab/conformance_queue.py": "d1ccf1d9d73cf703caf218324ca5f03767910e25ef1e73e903f23884426cdf95", + "src/qwen_r9700_lab/conformance_radiance.py": "d4980810ce03b2aa451738f321ba4346e5b8192f3034605b08674ab3888568a3", + "src/qwen_r9700_lab/conformance_reference.py": "0bc2e74d1295d53d839c822661e42ce8bddd36e63e36853418e9544187aa1e04", + "src/qwen_r9700_lab/conformance_reference_store.py": "720258558c8152c2f8e99035926a3abee95c111359c1a323e7da9098ba3d10b1", + "src/qwen_r9700_lab/conformance_replay.py": "178fcf75d2e88979c34085fba110df1df25277bbd30af543c905e109ce3a3834", + "src/qwen_r9700_lab/conformance_rotary_intervention.py": "a4ec7ce8cfebb0d715c5c259f6eb092dcf8854aeeee42e01a7f54c5988a19c57", + "src/qwen_r9700_lab/conformance_rotary_repair.py": "98f464b0dd53c2bbb44a4cc9e9307d520202a81a9a3175a3c091c8d95da050a0", + "src/qwen_r9700_lab/conformance_runtime.py": "830e09e70d609c28abe3e23fa13b6cb96176c70cb5aeba0da419029338d967ef", + "src/qwen_r9700_lab/conformance_scenarios.py": "57d5f84bd1db875a13f3c61fe65f5118de388e2736bb0e865558ee624472351f", + "src/qwen_r9700_lab/conformance_session.py": "3069b56e1af18dbc5db365968807459e45936f67ae0d1f8731250b02450b81b6", + "src/qwen_r9700_lab/conformance_shm.py": "eab5f933b6ce3be6f0942713230699d31ecfb4a73101d5776a523aa774b47e8c", + "src/qwen_r9700_lab/conformance_stage_isolation.py": "312a7e8353530e2807b89ecc5111a7de5bce3cc2dd56f850dd662490fd860149", + "src/qwen_r9700_lab/conformance_state.py": "42ade68e94bf53594b53b17ee7e916a77c443c18fe4b90a795e7ce510ac38c69", + "src/qwen_r9700_lab/conformance_topk.py": "940edebf21f2e8dd335ad52d37d7507d86a41e7fafc8884a5499e704069f30ce", + "src/qwen_r9700_lab/conformance_transport.py": "2bcf4eba295ea245c68d4f8bb8af28b68e60397086df72435ebd11973079cdef", + "src/qwen_r9700_lab/diagnostic_contract.py": "603680020b6a69d728d615eef9815b424610b4979410f223ab76e28dc4ee779e", + "src/qwen_r9700_lab/exact_fp8_metrics.py": "6c7aacb3a50e0f17c3f4eafdcd90b64b5efab8a63c9e0e8e57be98431d94a2bd", + "src/qwen_r9700_lab/manifests.py": "1eaea2743bcd851d54e6fc4809b5a26e9ab33a755fdbf547264183fc70ec6485", + "src/qwen_r9700_lab/ordered_reference_linear.c": "314ff8a91cd06dd14eafd5ee5695a2d90f77c5186820797a4460cb182fb3d272", + "src/qwen_r9700_lab/ordered_reference_linear.py": "7a8d5559918a3a21f2f0491f26942455f14aa0fe71b8a6029d4fcb80c4319bc7", + "src/qwen_r9700_lab/radiance_cache.py": "04adc11af696ab006570ee481b9e0b04d673e4bc6282469ea03061e6567abe03", + "tests/test_conformance_execution_modes.py": "cf4d3d64365c7797460f02a152bb43f50875a9ef3f79a1a4bed1d4f8d02c98b6", + "tests/test_conformance_rotary_repair.py": "0d5b771d005e805f4f3f0251ed82aab6ad1c39c0182f0811c7614de3f69e312b", + "tests/test_conformance_stage_isolation.py": "08e448a1620959e153907ad4b9d8d119a65b50990430946aebabab34c9c9df63", + "tests/test_d7_precision_comparison.py": "b40d22a9f2f27c63d0933b2f7a7e26f5804a3f9a406a2e28c5743c2032972902", + "tests/test_d7_rotary_comparison.py": "c78dd032b9aa9173da36fe5e16841f454207a8ca806065e6225d36d8c3a62e03" + }, + "origin": "https://github.com/Terrydaktal/qwen-r9700-lab", + "report": "https://github.com/Terrydaktal/d7-rdna4-report/blob/main/reports/d7-rdna4-2026-09-17/REPORT.md" +} diff --git a/benchmarks/conformance/configs/profiles/gdn-stock-m1-arithmetic-v1.json b/benchmarks/conformance/configs/profiles/gdn-stock-m1-arithmetic-v1.json new file mode 100644 index 0000000..bd6ceae --- /dev/null +++ b/benchmarks/conformance/configs/profiles/gdn-stock-m1-arithmetic-v1.json @@ -0,0 +1,152 @@ +{ + "schema": "urn:qwen:gdn-arithmetic-contract:v1", + "id": "stock-packed-gdn-m1-gfx1201-v1", + "scope": "GDN core from raw convolution Q/K/V and gate projections to output plus recurrent state; other operators have separate contracts.", + "status": "DECLARED; native cross-mode equivalence UNPROVED", + "selection": "Pinned stock vLLM packed single-token GDN, applied causally to each materialized token. R4D and NumPy remain separately identified implementations.", + "domain": { + "query_heads": 16, + "value_heads": 48, + "key_width": 128, + "value_width": 128, + "batch_sequences": 1, + "qkv_width": 10240, + "state_order": [ + "head", + "value", + "key" + ], + "state_elements": 786432, + "inputs": "Finite BF16 Q/K/V, a, b, dt_bias; finite FP32 A_log and initial state. Invalid geometry, aliasing, nonfinite output/state or incomplete evidence fails closed." + }, + "precision": { + "qkv": "BF16 widened exactly to FP32", + "a_b_dt_bias": "BF16 widened exactly to FP32", + "A_log": "FP32", + "qk_normalization": "FP32 throughout; no intermediate BF16 cast", + "beta": "sigmoid in the pinned FP32 kernel, then BF16 round-to-nearest ties-even, then exact FP32 widening", + "decay_gate": "FP32", + "recurrent_state": "FP32 on every read, update and committed write", + "output": "BF16 round-to-nearest ties-even" + }, + "operations": [ + { + "name": "normalize", + "formula": "n(x) = x / sqrt(sum(x_i*x_i) + float32(1e-6))", + "arithmetic": "Pinned compiled reduction, square root and division. No replacement with a differently rounded reciprocal-square-root multiply." + }, + { + "name": "scale_query", + "formula": "q = float32(n(Q) * float32(128**-0.5)); k = n(K)", + "arithmetic": "Scale q before the output dot product; not after it." + }, + { + "name": "gates", + "formula": "x = float32(a + dt_bias); softplus = x if x > 20 else log(1 + exp(x)); g = -exp(A_log)*softplus; beta = widen(BF16_RNE(sigmoid(b)))", + "arithmetic": "The pinned kernel intrinsics and their compiled implementations define rounding. No cumulative-gate reconstruction or future-dependent fallback." + }, + { + "name": "decay", + "formula": "D = float32(exp(g)*S)" + }, + { + "name": "prediction", + "formula": "p = dot_stock(D,k)" + }, + { + "name": "residual", + "formula": "r = float32(float32(v-p)*beta)" + }, + { + "name": "state", + "formula": "S_next = update_stock(D,r,k)", + "arithmetic": "Retain the compiled multiply-add contraction; separate multiply/add is a different numerical path." + }, + { + "name": "output", + "formula": "o = BF16_RNE(dot_stock(S_next,q))" + } + ], + "arithmetic_authority": { + "kind": "pinned-native-executable", + "note": "Real-number formulae describe dataflow. Exact finite-precision results are defined by the retained kernel artifact and packing below, under the declared compiler/hardware assumptions. Source inspection or matching hashes is not proof of runtime dispatch or functional correctness.", + "kernel": "fused_recurrent_gated_delta_rule_packed_decode_kernel", + "source": { + "fused_recurrent.py": "00a3b971b0dbb6ed26e246970a0e1a21a9a174030974685bdd3f0c8ab5fb4bfe", + "fla_op.py": "456da84fd12411cf53c96a90dd4f78f49afb454c3d51d2c524ea8e7106fe8ae5" + }, + "cache_entry": "XW6RXQHLHKDBX2FF2LE7HICYZGQCSR5L4HJZ7USZF4L2FZQULJTQ", + "files": { + "fused_recurrent_gated_delta_rule_packed_decode_kernel.source": "f47615b4e3c0a4136cb3d5f7e526f222c8609e73dff2d89111c34ba1745ac1a7", + "fused_recurrent_gated_delta_rule_packed_decode_kernel.ttir": "0333349d4ff544078a5d35354a5ca42d7fe37360af7ce188d9778dba7887aabf", + "fused_recurrent_gated_delta_rule_packed_decode_kernel.ttgir": "889064969acdafd8931d3d7ae02e5b0ce98fdb80ac1f9df9fb919b5259c2645c", + "fused_recurrent_gated_delta_rule_packed_decode_kernel.llir": "eb6f3975ed77e9f0b53975d1d1c0739abdba06813e64bea8aa6f66baeadd19f2", + "fused_recurrent_gated_delta_rule_packed_decode_kernel.amdgcn": "f07d71ae362c0e87633253a85c2571e0ac626ff8b520034e16b554687ec4e1a1", + "fused_recurrent_gated_delta_rule_packed_decode_kernel.hsaco": "9feeb930b8635122c552ad727fbe897d0854332f9a7880468662b9ed83350135", + "fused_recurrent_gated_delta_rule_packed_decode_kernel.json": "de1084d3b9e844ded01cd70f866eda11618a6e95e1471b0ccc0f295921c96fad" + }, + "target": { + "backend": "hip", + "arch": "gfx1201", + "warp_size": 32 + }, + "compiler": { + "triton_version": "3.7.1", + "num_warps": 1, + "num_stages": 3, + "enable_fp_fusion": true, + "allow_flush_denorm": false, + "sanitize_overflow": true + }, + "environment": { + "FLA_USE_FAST_OPS": "0" + }, + "packing": { + "query_heads": 16, + "value_heads": 48, + "K": 128, + "V": 128, + "BK": 128, + "BV": 32, + "state_token_stride": 802816, + "a_token_stride": 96, + "b_token_stride": 96, + "index_stride": 1, + "state_slot": 1, + "grid": [ + 4, + 48 + ], + "scale_fp32_hex": "f304b53d", + "note": "Single row makes the packed-QKV row stride irrelevant to this invocation. Reject or separately qualify other physical packing; physical block numbers are not semantic state." + } + }, + "execution_modes": { + "m1": "One invocation F(S,u) with exactly one materialized input token.", + "prefill": "Ordered fold of that same F over tokens. Empty segment is the identity. Every intermediate prefix must match.", + "d7": "Verification rows are u[0]=previously emitted pending token, u[1:8]=seven proposals. For accepted proposal count k in 0..7, commit the state after rows 0..k (k+1 processed rows); the newly emitted correction/bonus token stays pending.", + "rollback": "Rejected rows cannot affect committed GDN state, convolution history, KV, positions or events. The GDN-only contract does not certify those other components.", + "snapshot": "Persist the contract identity with state; do not treat states created under a different arithmetic contract as interchangeable without checking or rebuilding." + }, + "obligations": { + "partition": "fold(F,S,A+B) == fold(F,fold(F,S,A).state,B) for outputs and state, including empty segments", + "causal_prefix": "Changing inputs after prefix p leaves every output/state through p byte-identical.", + "accepted_prefix": "D7_commit(S,u,k) == fold(F,S,u[:k+1]).state with a separate pending-token field.", + "exactness": "Compare BF16 output bits and every FP32 logical-state bit; hashes and error tolerances alone do not establish equality." + }, + "separate_targets": { + "numpy-serial-fp32-v1": "Independent canonical model. It rounds normalized Q/K to BF16, uses ordered non-fused reductions and scales the final output dot. It is not relabelled as this stock kernel.", + "r4d-unfused-m1": "Retains FP32 beta, uses its own reduction/normalization intrinsics, disables FMA and rounds output ties away. Existing prefill candidate targets this path, not this contract.", + "huggingface-fallback": "The pinned Transformers fallback also returns BF16 beta, but its normalization and framework reductions differ. Full bit equality with it is UNPROVED." + }, + "required_qualification": [ + "Reproduce the preserved stock first transition exactly with the selected invocation packing.", + "Compare repaired prefill/M1/D7 against this oracle from equal independent initial state.", + "Exercise every accepted width, all-prefix equality, future-suffix mutations, padding, nonzero state, extreme gates and BF16 halfway boundaries.", + "Then qualify graph execution, full-model logits/state, snapshots and long contexts; benchmark separately." + ], + "production_changed": false, + "existing_matrix_contract_changed": false, + "native_repair_qualified": false, + "sha256": "d8149b00210c94acfca4f1f66030de302dcaf8155894de6a23d3f4b76da61e00" +} diff --git a/benchmarks/conformance/pyproject.toml b/benchmarks/conformance/pyproject.toml new file mode 100644 index 0000000..3b6c5f7 --- /dev/null +++ b/benchmarks/conformance/pyproject.toml @@ -0,0 +1,38 @@ +[project] +name = "radiance-conformance" +version = "0.1.0" +requires-python = ">=3.12" +dependencies = [ + "numpy>=2", + "jsonschema>=4.23,<5", + "PyYAML>=6", + "psutil>=6", + "requests>=2", + "safetensors>=0.5", +] + +[project.optional-dependencies] +cpu-tests = [ + "torch>=2.14.0", +] + +[dependency-groups] +dev = ["pytest>=8", "z3-solver>=4.13"] + +[build-system] +requires = ["hatchling"] +build-backend = "hatchling.build" + +[tool.hatch.build.targets.wheel] +packages = ["src/qwen_r9700_lab"] + +[[tool.uv.index]] +name = "pytorch-cpu" +url = "https://download.pytorch.org/whl/cpu" +explicit = true + +[tool.pytest.ini_options] +pythonpath = ["src", "tests"] + +[tool.uv.sources] +torch = { index = "pytorch-cpu" } diff --git a/benchmarks/conformance/src/qwen_r9700_lab/__init__.py b/benchmarks/conformance/src/qwen_r9700_lab/__init__.py new file mode 100644 index 0000000..5646d7c --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/__init__.py @@ -0,0 +1 @@ +"""Source-bound conformance helpers; importing them does not activate inference.""" diff --git a/benchmarks/conformance/src/qwen_r9700_lab/config.py b/benchmarks/conformance/src/qwen_r9700_lab/config.py new file mode 100644 index 0000000..61e119d --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/config.py @@ -0,0 +1,228 @@ +"""Configuration discovery and validation.""" + +from __future__ import annotations + +import json +import os +from collections.abc import Iterable +from pathlib import Path +from typing import Any + +from jsonschema import Draft202012Validator, FormatChecker + +SCHEMA_FILENAMES = { + "engine": "engine.config.schema.json", + "hardware": "hardware.config.schema.json", + "model": "model.config.schema.json", + "profile": "profile.config.schema.json", +} + + +class LabError(RuntimeError): + """Base class for expected, user-facing errors.""" + + +class ConfigurationError(LabError): + """A configuration or manifest did not satisfy its contract.""" + + +def find_project_root(start: Path | None = None) -> Path: + """Find a source checkout containing the configuration and schemas.""" + + configured = os.environ.get("QWEN_R9700_LAB_ROOT") + roots: list[Path] = [] + if configured: + roots.append(Path(configured).expanduser()) + + for origin in (start or Path.cwd(), Path(__file__).resolve()): + candidate = origin if origin.is_dir() else origin.parent + roots.extend((candidate, *candidate.parents)) + + for root in roots: + if (root / "schemas").is_dir() and (root / "configs").is_dir(): + return root.resolve() + raise ConfigurationError( + "could not locate the project root; run from the checkout or set QWEN_R9700_LAB_ROOT" + ) + + +def load_json_object(path: Path) -> dict[str, Any]: + """Load a JSON object with concise diagnostics.""" + + try: + with path.open(encoding="utf-8") as handle: + value = json.load(handle) + except FileNotFoundError as error: + raise ConfigurationError(f"file does not exist: {path}") from error + except json.JSONDecodeError as error: + raise ConfigurationError( + f"invalid JSON in {path} at line {error.lineno}, column {error.colno}: {error.msg}" + ) from error + except OSError as error: + raise ConfigurationError(f"could not read {path}: {error}") from error + + if not isinstance(value, dict): + raise ConfigurationError(f"expected a JSON object in {path}") + return value + + +def validate_instance(instance: dict[str, Any], schema_path: Path, label: str) -> None: + """Validate an instance and report all schema violations together.""" + + schema = load_json_object(schema_path) + try: + Draft202012Validator.check_schema(schema) + except Exception as error: # pragma: no cover - indicates a repository defect + raise ConfigurationError(f"invalid repository schema {schema_path}: {error}") from error + + errors = sorted( + Draft202012Validator(schema, format_checker=FormatChecker()).iter_errors(instance), + key=lambda error: tuple(str(part) for part in error.absolute_path), + ) + if not errors: + return + + lines = [] + for error in errors: + location = ".".join(str(part) for part in error.absolute_path) or "" + lines.append(f"{location}: {error.message}") + raise ConfigurationError(f"{label} failed validation:\n " + "\n ".join(lines)) + + +def validate_config(path: Path, root: Path | None = None) -> dict[str, Any]: + """Validate one typed repository configuration file.""" + + project_root = root or find_project_root(path.parent) + instance = load_json_object(path) + kind = instance.get("kind") + if kind not in SCHEMA_FILENAMES: + expected = ", ".join(sorted(SCHEMA_FILENAMES)) + raise ConfigurationError( + f"{path}: unknown config kind {kind!r}; expected one of {expected}" + ) + + validate_instance(instance, project_root / "schemas" / SCHEMA_FILENAMES[kind], str(path)) + if kind == "profile": + validate_profile_semantics(instance, path) + return instance + + +def validate_profile_semantics(profile: dict[str, Any], path: Path) -> None: + """Enforce context/RoPE invariants that are clearer in code than JSON Schema.""" + + maximum = profile["max_context_tokens"] + native = profile["native_context_tokens"] + qualification = profile["backend_qualification"] + status = qualification["status"] + rope = profile["rope"] + kv_cache = profile["kv_cache"] + speculation = profile["speculation"] + + if maximum <= native: + if rope["type"] != "native": + raise ConfigurationError(f"{path}: native context profiles must not enable YaRN") + if "factor" in rope: + raise ConfigurationError(f"{path}: native RoPE must not declare a scaling factor") + else: + if rope["type"] != "yarn": + raise ConfigurationError(f"{path}: extended context profiles must use YaRN") + minimum_factor = maximum / native + if rope["factor"] + 1e-9 < minimum_factor: + raise ConfigurationError( + f"{path}: YaRN factor {rope['factor']} is too small for {maximum} tokens " + f"(minimum {minimum_factor:.3f})" + ) + if rope["original_max_position_embeddings"] != native: + raise ConfigurationError( + f"{path}: YaRN original_max_position_embeddings must equal " + f"native_context_tokens ({native})" + ) + + if qualification["target_only"] and ( + speculation["preferred_drafter"] != "none" + or speculation["maximum_draft_tokens"] != 0 + or speculation["adaptive"] + ): + raise ConfigurationError( + f"{path}: target-only backend qualifications must disable speculation" + ) + + quality_gates = qualification["quality_gates"] + if status == "unavailable": + if kv_cache["representation"] != "unavailable": + raise ConfigurationError( + f"{path}: unavailable profiles must not claim an executable KV representation" + ) + if not quality_gates: + raise ConfigurationError( + f"{path}: unavailable profiles must state the research gates that block execution" + ) + return + + if kv_cache["representation"] == "unavailable": + raise ConfigurationError( + f"{path}: executable profiles must declare an implemented KV representation" + ) + if kv_cache["prefix_caching"] != (kv_cache["mamba_cache_mode"] == "align"): + raise ConfigurationError( + f"{path}: prefix caching requires mamba-cache-mode align; disabled prefix caching " + "requires mode none" + ) + + if status == "production_ready": + if maximum > 131072: + raise ConfigurationError( + f"{path}: stock-vLLM production qualification stops at 131072 tokens" + ) + if kv_cache["representation"] != "bfloat16" or kv_cache["cli_dtype"] != "auto": + raise ConfigurationError( + f"{path}: production-ready profiles must use BF16 storage via KV dtype auto" + ) + if quality_gates: + raise ConfigurationError( + f"{path}: production-ready profiles cannot retain unresolved quality gates" + ) + if maximum >= 65536 and not kv_cache["prefix_caching"]: + raise ConfigurationError( + f"{path}: 64K and 128K production profiles require aligned prefix caching" + ) + elif status == "experimental_quality_gate": + if maximum != native: + raise ConfigurationError( + f"{path}: the current experimental qualification is only for native 262144 context" + ) + if kv_cache["representation"] != "fp8_e4m3" or kv_cache["cli_dtype"] != "fp8_e4m3": + raise ConfigurationError( + f"{path}: the 262144-token experiment must declare FP8 E4M3 explicitly" + ) + if not quality_gates: + raise ConfigurationError( + f"{path}: experimental profiles must state their unresolved quality gates" + ) + + +def validate_profiles(paths: Iterable[Path], root: Path | None = None) -> list[dict[str, Any]]: + """Validate profiles individually and enforce uniqueness across a set.""" + + path_list = list(paths) + if not path_list: + raise ConfigurationError("no profile configuration files were selected") + + profiles: list[dict[str, Any]] = [] + names: dict[str, Path] = {} + for path in path_list: + profile = validate_config(path, root) + if profile["kind"] != "profile": + raise ConfigurationError(f"{path}: expected kind 'profile'") + name = profile["name"] + if name in names: + raise ConfigurationError(f"duplicate profile name {name!r}: {names[name]} and {path}") + names[name] = path + profiles.append(profile) + return profiles + + +def all_config_paths(root: Path) -> list[Path]: + """Return every checked-in configuration in stable order.""" + + return sorted((root / "configs").glob("*/*.json")) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_artifacts.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_artifacts.py new file mode 100644 index 0000000..cdec243 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_artifacts.py @@ -0,0 +1,168 @@ +"""Content-free runtime identity, with unobserved device artifacts explicit. + +Reading process mappings and library files does not execute GPU code. A mapped +file's hash is not proof of the actual device instructions dispatched from it. +""" + +from __future__ import annotations + +import hashlib +import importlib.metadata +import os +import platform +import sys +from functools import lru_cache +from pathlib import Path + +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, seal + +FLAGS = ( + "VLLM_USE_V2_MODEL_RUNNER", + "VLLM_ATTENTION_BACKEND", + "VLLM_ROCM_USE_AITER", + "RADIANCE_MXFP4", + "RADIANCE_MXFP4_W4A8", + "RADIANCE_MXFP4_WPERM", + "RADIANCE_MXFP4_DECODE_NT", + "RADIANCE_MXFP4_DECODE_MAX_M", + "RADIANCE_MXFP4_W4A8_MIN_M", + "RADIANCE_MXFP4_TN4_MIN_M", + "RADIANCE_VERIFY_HEAD", + "RADIANCE_VERIFY_HEAD_TOPK", + "RADIANCE_DYNAMIC_SPEC_WIDTH", + "RADIANCE_NORMQUANT_FUSION", + "RADIANCE_FP8_STREAM", + "RADIANCE_TILED_PREFILL", + "RADIANCE_GDN_NORMQUANT_FUSION", + "RADIANCE_PREFILL_FP8", + "GPU_MAX_HW_QUEUES", + "HSA_ENABLE_MWAITX", + "TORCHINDUCTOR_EMULATE_PRECISION_CASTS", +) + + +def file_identity(path): + """Hash bytes from one stable file descriptor; never substitute a guessed hash.""" + h = hashlib.sha256() + with path.open("rb") as stream: + before = os.fstat(stream.fileno()) + while chunk := stream.read(1024 * 1024): + h.update(chunk) + after = os.fstat(stream.fileno()) + fields = ("st_dev", "st_ino", "st_size", "st_mtime_ns", "st_ctime_ns") + if any(getattr(before, f) != getattr(after, f) for f in fields): + raise DiagnosticError("runtime artifact changed during hashing") + return {"sha256": h.hexdigest(), "bytes": after.st_size} + + +def mapped_libraries(text): + paths = set() + for line in text.splitlines(): + fields = line.split(maxsplit=5) + if len(fields) == 6 and fields[5].startswith("/"): + path = fields[5] + if ".so" in Path(path).name or path.endswith((".hsaco", ".co")): + paths.add(path) + return sorted(paths) + + +def compiler_settings(modules=None): + """Observe an already-loaded compiler; never import/initialize Torch here.""" + modules = sys.modules if modules is None else modules + config = modules.get("torch._inductor.config") + value = getattr(config, "emulate_precision_casts", None) if config is not None else None + return {"emulate_precision_casts": value if type(value) is bool else None} + + +def capture_runtime(*, maps_path=Path("/proc/self/maps"), environ=None): + """Observe library files and an explicit nonsecret environment allowlist.""" + env = os.environ if environ is None else environ + libraries, unavailable = {}, {} + try: + mapped = mapped_libraries(maps_path.read_text()) + except OSError as exc: + mapped = [] + unavailable["process_mappings"] = type(exc).__name__ + for name in mapped: + try: + libraries[name] = file_identity(Path(name)) + except (OSError, DiagnosticError) as exc: + unavailable[name] = type(exc).__name__ + versions = {} + for name in ("numpy", "vllm", "torch", "triton", "aiter", "libr4d"): + try: + versions[name] = importlib.metadata.version(name) + except importlib.metadata.PackageNotFoundError: + versions[name] = None + return seal( + { + "schema": "urn:qwen:conformance-runtime-artifacts:v1", + "python": sys.version, + "kernel": platform.release(), + "machine": platform.machine(), + "packages": versions, + "flags": {name: env.get(name) for name in FLAGS}, + "compiler_settings": compiler_settings(), + "mapped_file_bytes": libraries, + "unavailable_files": unavailable, + "unproved": { + "device_code_objects": "Mapped library hashes do not attest dispatched GPU ISA.", + "compiler_flags_and_generated_ir": "Requires compiler/dispatch instrumentation.", + "firmware_and_driver_execution": "Not independently verified by this collector.", + }, + "artifact_inventory_complete": False, + "exact_device_binary_attested": False, + } + ) + + +@lru_cache(maxsize=128) +def _cached_file_identity(path, stat_key): + # The key contains inode, timestamps and size. Replacements invalidate it. + return file_identity(Path(path)) + + +def reference_runtime_identity(): + """Bind the CPU oracle's executable dependencies, not just package versions. + + No GPU imports/queries. Loaded-file identity and CPU dispatch declarations + are evidence under a stable-process assumption, not a compiler proof. + """ + import math + + import numpy as np + + core = sys.modules["numpy._core._multiarray_umath"] + paths = {Path(sys.executable).resolve(), Path(core.__file__).resolve()} + if getattr(math, "__file__", None): + paths.add(Path(math.__file__).resolve()) + maps = Path("/proc/self/maps") + if not maps.is_file(): + raise DiagnosticError("CPU reference library mapping is unavailable") + for path in mapped_libraries(maps.read_text()): + name = Path(path).name + if name.startswith(("libm.so", "libc.so", "libpython", "ld-linux")): + paths.add(Path(path).resolve()) + libraries = {} + for path in sorted(paths): + info = path.stat() + key = (info.st_dev, info.st_ino, info.st_size, info.st_mtime_ns, info.st_ctime_ns) + libraries[str(path)] = dict(_cached_file_identity(str(path), key)) + return seal( + { + "schema": "urn:qwen:reference-cpu-runtime:v1", + "python": sys.version, + "numpy": np.__version__, + "machine": platform.machine(), + "byteorder": sys.byteorder, + "files": libraries, + "numpy_cpu_features": dict(core.__cpu_features__), + "assumptions": [ + ("loaded memory matches recorded files"), + ("NumPy/libm implement the declared arithmetic"), + ("no concurrent runtime patching"), + ], + "compiler_correctness": "UNPROVED", + "gpu_used": False, + } + ) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_attention_cut.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_attention_cut.py new file mode 100644 index 0000000..94e51f1 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_attention_cut.py @@ -0,0 +1,198 @@ +"""Locate Qwen attention sub-boundaries by owners, tensor roles and checked wiring.""" + +import numpy as np + +from qwen_r9700_lab.conformance_topk import require +from qwen_r9700_lab.diagnostic_contract import authenticate + + +def attention_cut(metadata, tensors, layer): + authenticate(metadata) + stem = f"language_model.model.layers.{layer}.self_attn." + events = metadata["events"] + rows = len(metadata["positions"]) + + def owned(part): + found = [ + e + for e in events + if any(o.startswith(stem + part + ".") for o in e.get("logical_identities", ())) + ] + require(len(found) == 1, "attention owner missing or ambiguous") + return found[0] + + qkv, out, qnorm, knorm = [owned(s) for s in ("qkv_proj", "o_proj", "q_norm", "k_norm")] + require( + qkv["operation"] == out["operation"] == "radiance.mxfp4_linear.default", + "unknown attention projection implementation", + ) + middle = [e for e in events if qkv["index"] < e["index"] < out["index"]] + attention = [ + e for e in middle if e["operation"] == "vllm.unified_attention_with_output.default" + ] + require(len(attention) == 1, "attention call missing or ambiguous between projections") + attention = attention[0] + require( + qkv["index"] < qnorm["index"] < knorm["index"] < attention["index"] < out["index"], + "unexpected attention cut order", + ) + tail = [e for e in middle if e["index"] > attention["index"]] + compiled = ( + len(tail) == 1 + and tail[0]["operation"] == "inductor/triton_poi_fused_mul_mxfp4_linear_sigmoid_view_0" + ) + eager = [e["operation"] for e in tail] == ["aten.sigmoid.default", "aten.mul.Tensor"] + require(compiled or eager, "unsupported attention gating sequence") + + def get(event, suffix, shape): + key = f"{event['index']}.{suffix}" + phase = suffix.split(".")[0] + found = [r for r in event[phase] if r["key"] == key] + require(len(found) == 1 and key in tensors, "missing attention tensor role") + value = tensors[key] + require( + found[0]["dtype"] == "torch.bfloat16" + and found[0]["shape"] == shape + and list(value.shape) == shape + and value.dtype == np.int16, + "attention cut needs the pinned BF16 logical layout", + ) + return value + + projection = get(qkv, "after.result", [rows, 14336]) + query_input = get(qnorm, "before.args.0", [rows, 24, 256]) + key_input = get(knorm, "before.args.0", [rows, 4, 256]) + query = get(qnorm, "after.result", [rows, 24, 256]) + key = get(knorm, "after.result", [rows, 4, 256]) + rotated_query = get(attention, "before.args.0", [rows, 24, 256]) + rotated_key = get(attention, "before.args.1", [rows, 4, 256]) + value = get(attention, "before.args.2", [rows, 4, 256]) + # args.3 is an output allocation before it has been written. Never compare + # its prior contents as if they were a semantically meaningful model input. + attended = get(attention, "after.mutable.output", [rows, 24, 256]) + gated = get(out, "before.args.0", [rows, 6144]) + if compiled: + gate = get(tail[0], "before.args.1", [rows, 6144]) + gate_attention = get(tail[0], "before.args.0", [rows, 24, 256]).reshape(rows, 6144) + gate_result = get(tail[0], "after.out_ptr0", [rows, 6144]) + else: + gate = get(tail[0], "before.args.0", [rows, 6144]) + gate_attention = get(tail[1], "before.args.0", [rows, 6144]) + gate_result = get(tail[1], "after.result", [rows, 6144]) + require( + np.array_equal( + get(tail[0], "after.result", [rows, 6144]), + get(tail[1], "before.args.1", [rows, 6144]), + ), + "sigmoid does not connect to attention multiplication", + ) + packed_qg = projection[:, :12288].reshape(rows, 24, 512) + for a, b in ( + (query_input, packed_qg[:, :, :256]), + (gate, packed_qg[:, :, 256:].reshape(rows, 6144)), + (key_input, projection[:, 12288:13312].reshape(rows, 4, 256)), + (value, projection[:, 13312:].reshape(rows, 4, 256)), + (gate_attention, attended.reshape(rows, 6144)), + (gated, gate_result), + ): + require(np.array_equal(a, b), "attention tensors do not follow the declared wiring") + return { + "qkv_projection": projection, + "query_after_normalization": query, + "key_after_normalization": key, + "query_after_rotation": rotated_query, + "key_after_rotation": rotated_key, + "value": value, + "attention_output": attended, + "gate_input": gate, + "gated_attention_output": gated, + } + + +def bf16_float(bits): + """Decode BF16 storage exactly; the arithmetic oracle never initializes a GPU.""" + require(bits.dtype == np.int16, "expected BF16 storage") + return (bits.view(np.uint16).astype(np.uint32) << 16).view(np.float32) + + +def round_bf16(value, mode): + """Explicit finite FP32 -> BF16 RNE/RTZ, including sign and tie-to-even.""" + require(mode in {"rne", "rtz"}, "unknown BF16 rounding mode") + value = np.ascontiguousarray(value, dtype=np.float32) + require(np.isfinite(value).all(), "non-finite arithmetic is outside this oracle") + bits = value.view(np.uint32) + if mode == "rne": + bits = bits + np.uint32(0x7FFF) + ((bits >> 16) & 1) + result = (bits >> 16).astype(np.uint16).view(np.int16) + require(np.isfinite(bf16_float(result)).all(), "BF16 overflow is outside this oracle") + return result + + +def rotary_formula(normalized, cosine, sine, product_rounding): + """NeoX pairs with declared product rounding, RNE addition and unchanged tail.""" + require( + normalized.ndim == 3 + and normalized.shape[2] == 256 + and normalized.shape[1] in {4, 24} + and cosine.shape == sine.shape == (normalized.shape[0], 32), + "unsupported pinned rotary layout", + ) + q, c, s = bf16_float(normalized), bf16_float(cosine)[:, None, :], bf16_float(sine)[:, None, :] + a, b = q[:, :, :32], q[:, :, 32:64] + + def product(x, y): + return bf16_float(round_bf16(x * y, product_rounding)) + + first = round_bf16(product(a, c) - product(b, s), "rne") + second = round_bf16(product(b, c) + product(a, s), "rne") + return np.concatenate((first, second, normalized[:, :, 64:]), axis=2) + + +def selected_rotary_coefficients(metadata, tensors, layer): + """Read compiled selected coefficients, checking their input wiring to both rotations.""" + authenticate(metadata) + owner = f"language_model.model.layers.{layer}.self_attn.rotary_emb.cos_sin_cache" + norm_owner = f"language_model.model.layers.{layer}.self_attn.k_norm." + norms = [ + e + for e in metadata["events"] + if any(o.startswith(norm_owner) for o in e.get("logical_identities", ())) + ] + require(len(norms) == 1, "missing unique K norm before rotary selection") + norm_index = norms[0]["index"] + consumers = [ + e + for e in metadata["events"] + if e["index"] > norm_index + and e["operation"] == "vllm.unified_attention_with_output.default" + ] + require(consumers, "no attention consumer follows rotary selection") + consumer_index = min(e["index"] for e in consumers) + # Cache storage may be shared by multiple layers. Restrict ownership to the + # validated K-norm -> attention interval, not every use of that allocation. + found = [ + e + for e in metadata["events"] + if norm_index < e["index"] < consumer_index and owner in e.get("logical_identities", ()) + ] + require(len(found) == 1, "missing unique compiled rotary coefficient selector") + selector = found[0] + require(selector["operation"].startswith("inductor/"), "coefficients need compiled capture") + positions = metadata["positions"] + coefficients = [tensors[f"{selector['index']}.after.out_ptr{i}"] for i in (0, 1)] + require( + all(x.shape == (len(positions), 32) and x.dtype == np.int16 for x in coefficients), + "unsupported coefficient layout", + ) + following = [ + e for e in metadata["events"] if selector["index"] < e["index"] <= selector["index"] + 2 + ] + require(len(following) == 2, "missing rotations after coefficient selection") + for event in following: + require(event["operation"].startswith("inductor/"), "unknown rotation implementation") + for i, expected in enumerate(coefficients, 1): + require( + np.array_equal(tensors[f"{event['index']}.before.args.{i}"], expected), + "rotation uses different coefficient inputs", + ) + return coefficients diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_boundaries.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_boundaries.py new file mode 100644 index 0000000..747cf98 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_boundaries.py @@ -0,0 +1,223 @@ +"""Canonical semantic-boundary observations, reusable across runtime adapters.""" + +import re +from pathlib import Path + +import numpy as np + +from qwen_r9700_lab.conformance_state import FrameWriter, compare_frames, read_frame +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + private_json, + seal, + write_private, +) + +STAGES = ("input_norm", "post_attention_norm", "output") + + +def validate_domain(positions, layers, stages): + if ( + not isinstance(positions, list) + or not positions + or any(type(p) is not int or p < 0 for p in positions) + or sorted(set(positions)) != positions + or type(layers) is not int + or layers < 1 + or not isinstance(stages, list) + or len(stages) != layers + ): + raise DiagnosticError("invalid semantic observation domain") + for row in stages: + if ( + not isinstance(row, list) + or not row + or any( + not isinstance(s, str) or not re.fullmatch(r"[A-Za-z][A-Za-z0-9_.-]*", s) + for s in row + ) + or len(set(row)) != len(row) + ): + raise DiagnosticError("invalid or duplicate semantic stage") + + +def detailed_stages(config): + """Native attention semantics; no inherited Quest selector requirement.""" + start = ["input", "input_norm"] + end = [ + "attention_residual", + "post_attention_norm", + "mlp_gate", + "mlp_up", + "mlp_intermediate", + "mlp_down", + "output", + ] + linear = [ + "gdn_qkv", + "gdn_z", + "gdn_b", + "gdn_a", + "conv_input_state", + "conv_output", + "conv_state", + "gdn_q", + "gdn_k", + "gdn_v", + "gdn_decay", + "gdn_beta", + "gdn_input_state", + "gdn_output", + "gdn_state", + "gdn_gated", + "attention_projected", + ] + attention = [ + "attention_q_projected", + "attention_k_projected", + "attention_v_projected", + "attention_q_norm", + "attention_k_norm", + "attention_q_rope", + "attention_k_rope", + "attention_key_stored", + "attention_value_stored", + "attention_output", + "attention_projected", + ] + return [ + start + (linear if k == "linear_attention" else attention) + end + for k in config["layer_types"] + ] + + +class BoundaryRecorder: + def __init__( + self, + root: Path, + *, + contract: str, + execution: str, + adapter: str, + positions: list[int], + layers: int, + input_digests: dict[int, str], + layer_stages=None, + ): + root.mkdir(mode=0o700) + self.root, self.positions, self.layers = root, positions, layers + self.identities = {"contract": contract, "execution": execution, "adapter": adapter} + self.inputs, self.recorded = input_digests, {} + self.stages = ( + [list(STAGES) for _ in range(layers)] if layer_stages is None else layer_stages + ) + validate_domain(positions, layers, self.stages) + self.position_set = frozenset(positions) + if ( + not positions + or len(set(positions)) != len(positions) + or layers < 1 + or len(self.stages) != layers + or any(not s or len(set(s)) != len(s) for s in self.stages) + or set(input_digests) != set(positions) + ): + raise DiagnosticError("invalid semantic observation domain") + + def record(self, position, layer, stage, value): + if position not in self.position_set or layer < 0: + return + if not 0 <= layer < self.layers: + raise DiagnosticError("unregistered semantic layer") + if stage not in self.stages[layer]: + return + key = (position, layer, stage) + if key in self.recorded: + raise DiagnosticError("semantic boundary was captured twice") + name = f"p{position:09d}-l{layer:03d}-{stage}" + writer = FrameWriter( + self.root / name, + **self.identities, + input_digest=self.inputs[position], + phase="operator", + consumed=position + 1, + pending=None, + expected=["value"], + logical={"layer": layer, "stage": stage}, + ) + writer.array("value", np.asarray(value, dtype="= value["case_timeout_seconds"]: + raise DiagnosticError("startup deadline must be shorter than the case deadline") + value.setdefault("seeds", [0, 17]) + if ( + not value["seeds"] + or len(set(value["seeds"])) != len(value["seeds"]) + or any(type(seed) is not int or not 0 <= seed < 2**31 for seed in value["seeds"]) + ): + raise DiagnosticError("campaign seeds must be explicit unique nonnegative integers") + default_contexts = [ + 8192, + 32768, + 60000, + 128000, + 200000, + value["max_context"] - value["output_tokens"] - 1, + ] + value.setdefault( + "contexts", + sorted( + { + n + for n in default_contexts + if n > 0 and n + value["output_tokens"] < value["max_context"] + } + ), + ) + if ( + not value["contexts"] + or len(set(value["contexts"])) != len(value["contexts"]) + or any( + type(n) is not int or n < 16 or n + value["output_tokens"] >= value["max_context"] + for n in value["contexts"] + ) + ): + raise DiagnosticError( + "context matrix is empty, duplicated or exceeds the configured window" + ) + if not isinstance(value["environment"], dict) or any( + not isinstance(k, str) or not isinstance(v, str) or "\0" in k + v + for k, v in value["environment"].items() + ): + raise DiagnosticError("campaign environment must contain string settings") + # No production destinations or externally supplied hooks from a caller's + # shell may redirect qualification into another session. + if any(k.startswith("QWEN_CONFORMANCE_") or k == "PYTHONPATH" for k in value["environment"]): + raise DiagnosticError("campaign environment cannot replace qualification wiring") + return value + + +def build_campaign(spec): + spec = validate_spec(spec) + cases = [] + + def add(family, variant, context=64, seed=0, **axes): + identity = f"{family}.{variant}.ctx{context}.seed{seed}" + cases.append( + { + "id": identity, + "family": family, + "variant": variant, + "context": context, + "seed": seed, + "axes": axes, + "oracle": FAMILIES[family], + } + ) + + for fault in FAULTS: + add("native_fault", fault) + # Independent serial CPU reference is deliberately limited to short native + # runs. Long contexts have separate self-consistency and operator oracles. + for context in (16, 63, 64, 65, 127, 128, 129): + for seed in spec["seeds"]: + add("forced_reference", "independent", context, seed) + boundaries = (63, 64, 65, 127, 128, 129, 1647, 1648, 1649, 2047, 2048, 2049, 8191, 8192, 8193) + for context in sorted( + set( + spec["contexts"] + + [n for n in boundaries if n + spec["output_tokens"] < spec["max_context"]] + ) + ): + for seed in spec["seeds"]: + for width in range(8): + add("forced_d7", f"accept{width}", context, seed, accepted=width) + for variant in ( + "speculation", + "verify_head", + "dynamic_width", + "graphs", + "async_experimental", + ): + add("natural", variant, context, seed) + add("natural", "head_omission_control", 129) + for width in range(7): + for seed in spec["seeds"]: + add("rejected_suffix", f"accept{width}", 129, seed, accepted=width) + for operator in ("mxfp4", "gdn", "norm_rope", "attention", "sampling"): + add("operator", operator) + add("operator", "native_dispatch", 256) + for context in spec["contexts"]: + for variant in ( + "warm", + "ram", + "eviction", + "clean_restart", + "shutdown_pending_tail", + "crash_restart", + "interrupted_write", + "corrupt_disk", + "missing_disk_block", + "compaction", + "cancellation", + ): + add("lifecycle", variant, context) + for variant in ( + "equal0", + "equal1", + "equal2", + "priority1_owner", + "priority1_waiter", + "priority2_waiter", + ): + add("priority", variant, context) + for variant in ( + "parser_fragments", + "pi_provider_fragments", + "tool_roundtrip", + "long_thinking", + "minimal_release", + "minimal_release_full_head", + "repeat_release_full_head", + "minimal_release_target_only", + "repeat_release_target_only", + "minimal_release_target_only_eager", + "repeat_release_target_only_eager", + ): + add("protocol", variant) + if len({case["id"] for case in cases}) != len(cases): + raise DiagnosticError("duplicate campaign case IDs") + cases.sort(key=case_priority) + for case in cases: + case["stage"] = ("pilot", "focused", "extended")[case_priority(case)[0]] + return seal( + { + "schema": SCHEMA, + "spec": spec, + "sources": source_identity(), + "cases": cases, + "universal_correctness": "UNPROVED", + "default_gpu_use": False, + "scope_limits": [ + "Finite inputs and interleavings; no claim about all arbitrary inputs.", + "Long-context M1/D7 agreement cannot exclude a bug shared by both native paths.", + "Operator tolerance tests are distinct from bit-exact reference equality.", + "Experimental async uses vLLM's scheduler, not the synchronous chat scheduler.", + "Graph entry observations do not prove GPU ISA or compiler correctness.", + "Tool/reasoning fixtures can reflect model errors; raw evidence is retained.", + ], + } + ) + + +def validate_campaign(campaign): + authenticate(campaign) + if campaign.get("schema") != SCHEMA: + raise DiagnosticError("unsupported campaign format") + canonical = build_campaign(campaign["spec"]) + if campaign != canonical: + raise DiagnosticError( + "campaign wiring, source identity or case inventory changed; re-plan explicitly" + ) + return campaign + + +def coverage(campaign, results): + validate_campaign(campaign) + expected = {case["id"]: case for case in campaign["cases"]} + if set(results) - set(expected): + raise DiagnosticError("result names a case outside the campaign") + rows = [] + for case in campaign["cases"]: + attempts = results.get(case["id"], []) + for result in attempts: + authenticate(result) + if ( + result.get("schema") != RESULT_SCHEMA + or result.get("campaign") != campaign["sha256"] + or result.get("case") != case + ): + raise DiagnosticError("case result belongs to different inputs, source or campaign") + if result.get("status") not in {"TESTED", "FAILED", "ERROR", "TIMEOUT", "UNSUPPORTED"}: + raise DiagnosticError("unsupported case status") + if result["status"] == "TESTED" and ( + result.get("executed") is not True or not result.get("checks") + ): + # CPU-only parser/provider cases still execute in the explicitly + # armed pinned qualification environment; their claim is scoped. + raise DiagnosticError("passing result is missing actual execution/oracle evidence") + status = ( + "NOT_RUN" + if not attempts + else ( + "TESTED" + if all(r["status"] == "TESTED" for r in attempts) + else next(r["status"] for r in attempts if r["status"] != "TESTED") + ) + ) + rows.append( + { + "id": case["id"], + "family": case["family"], + "status": status, + "attempts": [r["sha256"] for r in attempts], + "flaky": len(attempts) > 1 and len({r["status"] for r in attempts}) > 1, + } + ) + return seal( + { + "schema": "urn:qwen:conformance-coverage:v1", + "campaign": campaign["sha256"], + "cases": rows, + "counts": dict(Counter(row["status"] for row in rows)), + "complete": bool(rows) and all(row["status"] == "TESTED" for row in rows), + "universal_correctness": "UNPROVED", + "scope_limits": campaign["scope_limits"], + } + ) + + +def load_results(campaign, root): + results = {} + roots = [root] if isinstance(root, (str, Path)) else list(root) + roots = [Path(path).resolve() for path in roots] + if not roots or len(set(roots)) != len(roots): + raise DiagnosticError("result directories must be nonempty and unique") + for directory in roots: + if not directory.is_dir(): + raise DiagnosticError("campaign result directory is missing") + saved_campaign = private_json(directory / "campaign.json") + authenticate(saved_campaign) + if saved_campaign != campaign: + raise DiagnosticError("result directory belongs to another campaign") + for path in sorted( + directory.glob("case-*/result.json"), + key=lambda p: ( + p.parent.name.split("-attempt-")[0], + int(p.parent.name.split("-attempt-")[1]) if "-attempt-" in p.parent.name else 0, + ), + ): + result = private_json(path) + case = result.get("case", {}).get("id") + results.setdefault(case, []).append(result) + for attempts in results.values(): + attempts.sort(key=lambda result: result.get("finished_ns", 0)) + return results + + +def run_campaign( + campaign, + root, + *, + allow_gpu=False, + selected=None, + keep_going=False, + resume=False, + through="extended", + budget_seconds=None, + retry_failed=False, + min_free_bytes=20 * 1024**3, +): + if not allow_gpu: + raise DiagnosticError("GPU use was not authorized; campaign remains NOT_RUN") + validate_campaign(campaign) + selected = set([case["id"] for case in campaign["cases"]] if selected is None else selected) + if not selected or selected - {case["id"] for case in campaign["cases"]}: + raise DiagnosticError("empty selection or unknown campaign case") + spec = campaign["spec"] + if hashlib.sha256(Path(spec["python"]).read_bytes()).hexdigest() != spec["python_sha256"]: + raise DiagnosticError("qualification interpreter changed") + from qwen_r9700_lab.conformance_queue import run + + return run( + campaign, + root, + selected=selected, + keep_going=keep_going, + resume=resume, + through=through, + budget_seconds=budget_seconds, + retry_failed=retry_failed, + min_free_bytes=min_free_bytes, + ) + + +def execute_case(campaign, case, case_root): + spec = campaign["spec"] + started = time.monotonic() + try: + env = worker_environment(spec, case_root) + env.pop("QWEN_CONFORMANCE_NATIVE_EXPERIMENT", None) + env.pop("QWEN_CONFORMANCE_SERVER_SETTINGS", None) + process = OwnedProcess( + [spec["python"], "-m", "qwen_r9700_lab.conformance_campaign", str(case_root)], + case_root / "process", + env=env, + timeout=spec["case_timeout_seconds"], + ) + code = process.wait() + result = private_json(case_root / "worker-result.json") + if code and result.get("status") == "TESTED": + raise DiagnosticError("case process failed after claiming success") + return result + except Exception as error: + return case_result( + campaign, + case, + "TIMEOUT" if isinstance(error, TimeoutError) else "ERROR", + started=started, + error_type=type(error).__name__, + detail=str(error)[:1000], + ) + + +def case_result(campaign, case, status, *, started, checks=None, error_type=None, detail=None): + return seal( + { + "schema": RESULT_SCHEMA, + "campaign": campaign["sha256"], + "case": case, + "status": status, + "executed": status == "TESTED", + "gpu_executed": ( + False + if case["family"] == "protocol" + and case["variant"] in {"parser_fragments", "pi_provider_fragments"} + else (True if status == "TESTED" else None) + ), + "seconds": time.monotonic() - started, + "finished_ns": time.time_ns(), + "checks": checks or [], + "error_type": error_type, + "detail": detail, + "proof": "UNPROVED", + } + ) + + +def main(): + if os.environ.get("QWEN_CONFORMANCE_GPU") != "1": + raise DiagnosticError("case worker is not armed") + from qwen_r9700_lab.conformance_scenarios import SCENARIOS, UnavailableError + + root = Path(sys.argv[1]) + payload = private_json(root / "input.json") + campaign, case = payload["campaign"], payload["case"] + validate_campaign(campaign) + started = time.monotonic() + try: + verify_runtime_artifacts(campaign["spec"], root) + from contextlib import nullcontext + + from qwen_r9700_lab.conformance_gpu_lease import gpu_lease + + # The independent reference owns the GPU only for its native arm. + lease = ( + nullcontext() if case["family"] == "forced_reference" else gpu_lease(root / "gpu-lease") + ) + with lease: + checks = SCENARIOS[case["family"]](campaign["spec"], case, root) + if not checks: + raise DiagnosticError("case produced no observable checks") + result = case_result(campaign, case, "TESTED", started=started, checks=checks) + except Exception as error: + result = case_result( + campaign, + case, + "UNSUPPORTED" if isinstance(error, UnavailableError) else "FAILED", + started=started, + error_type=type(error).__name__, + detail=str(error)[:1000], + ) + write_private(root / "worker-result.json", result) + return 0 if result["status"] == "TESTED" else 1 + + +def verify_runtime_artifacts(spec, root): + """CPU-only source/package/binary checks before any inference library import.""" + import importlib.metadata + import importlib.util + + from qwen_r9700_lab.conformance_radiance import verify_sources + + module = importlib.util.find_spec("vllm") + if module is None or module.origin is None: + raise DiagnosticError("pinned Radiance runtime is unavailable") + package = Path(module.origin).resolve().parent.parent + verify_sources(package, spec["binding"]) + binary_hashes = spec["operator_profile"].get("kernel_hashes") + versions = spec["operator_profile"].get("package_versions") + if not binary_hashes or not versions: + raise DiagnosticError("campaign omitted its native binary or package identity") + for name, expected in binary_hashes.items(): + path = (package / name).resolve() + if ( + not path.is_relative_to(package) + or hashlib.sha256(path.read_bytes()).hexdigest() != expected + ): + raise DiagnosticError("native binary identity differs: " + name) + for name, expected in versions.items(): + if importlib.metadata.version(name) != expected: + raise DiagnosticError("native package identity differs: " + name) + write_private( + root / "runtime-identity.json", + { + "source_binding": spec["binding"]["sha256"], + "binary_hashes": binary_hashes, + "package_versions": versions, + "python_sha256": spec["python_sha256"], + "compiler_and_device_execution": "UNPROVED", + }, + ) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_cli.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_cli.py new file mode 100644 index 0000000..df100b2 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_cli.py @@ -0,0 +1,738 @@ +"""Offline conformance CLI; GPU use is a separate, explicitly armed command.""" + +from __future__ import annotations + +import argparse +import contextlib +import json +import os +import subprocess +import sys +import threading +from pathlib import Path + +from qwen_r9700_lab.conformance_artifacts import reference_runtime_identity +from qwen_r9700_lab.conformance_boundaries import compare_boundaries +from qwen_r9700_lab.conformance_control import audit_control +from qwen_r9700_lab.conformance_instrumentation import compare_calls +from qwen_r9700_lab.conformance_invariants import interval_certificate +from qwen_r9700_lab.conformance_lifecycle import compare_recovery +from qwen_r9700_lab.conformance_obligations import proof_obligations +from qwen_r9700_lab.conformance_proofs import run_obligations +from qwen_r9700_lab.conformance_reference import ( + REFERENCE_PROFILES, + reference_contract, + reference_semantics, +) +from qwen_r9700_lab.conformance_replay import ( + PLAN_SCHEMA, + reference_code_identity, + replay_operator, + run_reference, + validate_plan, +) +from qwen_r9700_lab.conformance_session import ( + CheckedSession, + SessionMismatchError, + compare_campaign, +) +from qwen_r9700_lab.conformance_state import compare_frames +from qwen_r9700_lab.diagnostic_contract import ( + ARTIFACT_GROUPS, + DiagnosticError, + authenticate, + digest, + private_json, + seal, + semantic_identity, + write_private, +) + +HELP = """NAME + qwen-conformance - exact state comparison and independent inference replay + +SYNOPSIS + qwen-conformance inventory + qwen-conformance proof-obligations [--profile NAME] [--output FILE] [--require-proved] + qwen-conformance make-plan --spec FILE --output FILE + qwen-conformance reference --plan FILE --output DIRECTORY + qwen-conformance native --plan FILE --config FILE --binding FILE --output DIRECTORY --allow-gpu + qwen-conformance compare --reference DIRECTORY --candidate DIRECTORY --output DIRECTORY + qwen-conformance boundaries --reference DIRECTORY --candidate DIRECTORY --output DIRECTORY + qwen-conformance calls --reference DIRECTORY --candidate DIRECTORY --output FILE + qwen-conformance recovery --plan FILE --reference DIRECTORY --live DIRECTORY + --restored DIRECTORY --output DIRECTORY + qwen-conformance control --trace FILE --output FILE + qwen-conformance certificate --bounds FILE --output FILE + qwen-conformance operator --capsule DIRECTORY --output DIRECTORY + qwen-conformance prove --output DIRECTORY + qwen-conformance suite-plan --spec FILE --output FILE + qwen-conformance suite-run --campaign FILE --output DIRECTORY --allow-gpu + [--case ID] [--keep-going] [--resume] [--through pilot|focused|extended] + [--budget-seconds N] [--retry-failed] [--min-free-gib N] + qwen-conformance suite-pause --output DIRECTORY + qwen-conformance suite-status --campaign FILE --results DIRECTORY [--results DIRECTORY] + [--require-passing] + qwen-conformance gate --authority DIRECTORY --transition FILE [--create] + +DESCRIPTION + Compares actual logical-state bytes, including independent initial prefill, + forced-token transitions and original quantized checkpoint operators. Finds + the first observed discrepancy without decoding or displaying private tokens. + Finite comparisons establish evidence for those observations, not a universal + proof. Missing observations and unsupported native layouts fail closed. + +OPTIONS + --allow-gpu Required for native and suite-run. Planning, comparison and + status commands use the CPU and never import GPU libraries. + --plan FILE Private, sealed execution plan from make-plan. + --spec FILE Private JSON: checkpoint path, checkpoint_files hashes, + kv_scales, prefix and forced_tokens; optional accepted_widths + reference_profile (default weight-only-bf16), and an explicit + observation_positions subset for a bounded tensor capture. Explicit + radiance-fp8 keeps the historical additional FP8 quantizers. + --config FILE Private native LLM kwargs; one eager TP1 worker, no connector. + --binding FILE Reviewed source binding for precisely one Radiance version. + --output PATH Private evidence destination; suite-run --resume continues it. + --create Create a new checked authority; otherwise resume its revision. + --trace FILE Complete sealed control event sequence, including source identity. + --bounds FILE Private full-vocabulary lower/upper rational bounds, winner, + and bound_origin. Bound soundness remains an explicit assumption. + --profile NAME Reference precision profile for the proof-obligations report. + --require-proved Return 1 while any whole-backend obligation remains unproved. + --campaign FILE Source-bound finite GPU campaign from suite-plan. + --case ID Select one case (repeatable); unselected cases remain NOT_RUN. + --keep-going Retain the first failure and continue independent cases. + --resume Continue the same checkpoint; passed cases are not rerun. + --through STAGE Run in priority order through pilot, focused or extended. + --budget-seconds N Stop between cases after this invocation's time budget. + --retry-failed Explicitly retry a reviewed failure; retain its earlier result. + --min-free-gib N Keep this free-space reserve plus estimated capture space (20). + --results DIR Existing campaign evidence (repeatable); retains all attempts. + --require-passing Return 1 for missing, failed, unsupported or unrun cases. + +OPERATION + make-plan -> independent reference/native captures -> compare/boundaries. + operator replays a captured numerical operation against the CPU reference. + prove checks actual small helper expressions, retaining SMT obligations. + proof-obligations lists the mathematical contract, component obligations, + implementation/test hashes and remaining gaps. It never upgrades tests to proofs. + recovery compares reference, live and restored logical state at the same prefix. + control checks completion, state versions, publication, ownership and snapshots. + calls compares captured module/operator inputs, mutations and outputs in causal order. + certificate checks the conditional argmax interval inequality for every token. + gate compares complete tentative state and output before an atomic revision + can be published. Tool events require separate checked protocol coverage. + Native replay is separate from the live Pi backend. It changes no live port, + snapshot, service or conversation. It needs an available GPU and is slow. + Full boundary capture includes every materialized prefill/accepted token and + can generate very large private evidence. Native BF16/FP8 extraction paths + exist but still require isolated GPU qualification before relying on them. + suite-plan wires fault injection, operator probes, forced and natural decode, + cache lifecycles, priority, graphs, parser and Pi-provider qualification. + suite-run launches only owned processes, loopback ports and private caches. + The complete inventory remains visible even when running a selected subset. + +EXAMPLES + qwen-conformance inventory + qwen-conformance prove --output /tmp/private-helper-proof + qwen-conformance compare --reference /tmp/ref --candidate /tmp/candidate --output /tmp/diff + +FILES + plan.json, frame.json and private tensor blobs: replay inputs and evidence. + schedule.json, boundaries.json: complete causal observation inventories. + authority.sqlite3: validated output revisions and retained state copies. + +PATHS + configs/profiles/radiance-conformance-v320260914.json: native source binding. + docs/backend-conformance.md: formats, admitted domain and GPU qualification plan. + +SECURITY NOTES + Evidence contains raw tokens and model state. Keep it on the authorized + machine, outside git, mode 0700 for directories and 0600 for files. + Native logs stay private. This is diagnostic process separation, not a + sandbox against malicious GPU kernels or another process with the same UID. + +EXIT STATUS + 0 Requested finite comparison/check completed successfully. + 1 A mismatch, failed helper obligation or undischarged --require-proved claim. + 2 Malformed evidence, missing coverage, unsupported path or worker failure. + +AUTHORS + Terrydaktal and contributors. +""" + + +def inventory(): + return { + "schema": "urn:qwen:conformance-readiness:v1", + "components": { + "independent_mxfp4_fp8_cpu_operators": "TESTED", + "tiny_hybrid_model_prefill_decode_restore": "TESTED", + "full_state_byte_comparator_and_fault_controls": "TESTED", + "durable_publication_gate": "TESTED", + "mapped_runtime_file_identity": "TESTED", + "d7_pending_and_rejected_suffix_helpers": "TESTED", + "compressed_store_ram_eviction_restart_recovery": "TESTED", + "three_way_recovery_corruption_localization": "TESTED", + "control_trace_faults_and_copy_on_write": "TESTED", + "source_bound_semantic_calls_and_cleanup": "TESTED", + "finite_native_campaign_wiring_and_missing_case_gate": "TESTED on CPU", + "owned_process_tree_deadlines_and_cleanup": "TESTED on CPU", + "native_lifecycle_priority_graph_head_and_protocol_campaign": "NOT_RUN on GPU", + "conditional_full_vocabulary_interval_checker": "TESTED; bound soundness ASSUMED", + "small_smt_helpers": "run prove for implementation-bound evidence", + "native_radiance_state_extractor": "UNPROVED", + "native_forced_m1_and_d7_replay": "UNPROVED", + "native_operator_and_boundary_equivalence": "UNPROVED", + "native_snapshot_restore_and_concurrency": "UNPROVED", + "sampled_distribution_and_tool_parser": "UNPROVED", + "compiled_graphs_and_async_races": "UNPROVED", + "compiler_and_hardware": "ASSUMED", + "actual_dispatched_device_binary_identity": "UNPROVED", + "whole_backend_formal_equivalence": "UNPROVED", + }, + "gpu_use_by_default": False, + "production_hooks_installed": False, + "reused_infrastructure": [ + "diagnostic_contract", + "conformance_gate", + "conformance_proofs", + "radiance layer/GDN capsules", + "verify-head observer", + "MXFP4/RMSNorm/R4D numerical probes", + "logical cache ownership validator", + ], + "references": {name: reference_contract(name) for name in REFERENCE_PROFILES}, + "default_new_plan_profile": "weight-only-bf16", + "whole_stack_obligations": proof_obligations(), + } + + +def make_plan(spec): + from qwen_r9700_lab.conformance_model import Checkpoint + + allowed = { + "checkpoint", + "checkpoint_files", + "kv_scales", + "prefix", + "forced_tokens", + "accepted_widths", + "reference_profile", + "observation_positions", + "reference_linear", + } + if set(spec) - allowed or allowed - { + "accepted_widths", + "reference_profile", + "observation_positions", + "reference_linear", + } - set(spec): + raise DiagnosticError("replay specification has missing or unsupported fields") + # Hash the actual checkpoint before naming it in the reference contract. + weights = Checkpoint(Path(spec["checkpoint"]), spec["checkpoint_files"]) + try: + config = weights.config.get("text_config", weights.config) + profile = spec.get("reference_profile", "weight-only-bf16") + arithmetic = reference_contract(profile) + semantics = reference_semantics( + spec["checkpoint_files"], config, spec["kv_scales"], profile + ) + code = reference_code_identity() + runtime = reference_runtime_identity() + execution = seal( + { + "schema": "urn:qwen:conformance-reference-execution:v1", + "reference_code": code, + "reference_runtime": runtime, + "reference_linear": spec.get("reference_linear"), + "semantics": semantic_identity(semantics), + "python": sys.version, + "unavailable": { + group: "native run must record this artifact group" + for group in sorted(ARTIFACT_GROUPS) + if group not in {"reference", "model"} + }, + "complete_attestation": False, + } + ) + plan = validate_plan( + seal( + { + "schema": PLAN_SCHEMA, + **spec, + "contract": semantic_identity(semantics), + "execution": execution["sha256"], + "adapter": digest(code), + "reference_arithmetic": arithmetic, + "reference_profile": profile, + "reference_semantics": semantics, + "reference_runtime": runtime["sha256"], + } + ) + ) + return plan, seal(semantics), execution + finally: + weights.close() + + +@contextlib.contextmanager +def reject_dead_native_rpcs(client): + """Fail pending diagnostic RPCs when the pinned sync transport loses its worker. + + Its output thread forwards engine death to generation consumers but leaves + utility Futures pending. This observer changes no live engine work and adds + no inference deadline: it only wakes waiters after declared engine death. + """ + from concurrent.futures import InvalidStateError + + resources, pending = client.resources, client.utility_results + stopped = threading.Event() + + def watch(): + while not stopped.wait(0.1): + if resources.engine_dead: + for future in list(pending.values()): + with contextlib.suppress(InvalidStateError): + future.set_exception(DiagnosticError("native worker exited during RPC")) + + thread = threading.Thread(target=watch, name="conformance-worker-death", daemon=True) + thread.start() + try: + yield + finally: + stopped.set() + thread.join() + + +def native_worker(plan_path, config_path, binding_path, root): + if os.environ.get("QWEN_CONFORMANCE_GPU") != "1": + raise DiagnosticError("GPU worker is not armed") + plan = validate_plan(private_json(plan_path)) + kwargs = private_json(config_path) + if ( + kwargs.get("model") != plan["checkpoint"] + or kwargs.get("enforce_eager") is not True + or kwargs.get("max_num_seqs") != 1 + ): + raise DiagnosticError("native config must match the plan and request one eager sequence") + if kwargs.get("kv_transfer_config") is not None or kwargs.get("async_scheduling", False): + raise DiagnosticError( + "native replay cannot use shared snapshots or asynchronous scheduling" + ) + if kwargs.get("tensor_parallel_size", 1) != 1 or kwargs.get("pipeline_parallel_size", 1) != 1: + raise DiagnosticError("native replay admits one GPU only") + if os.environ.get("VLLM_USE_V2_MODEL_RUNNER") != "1": + raise DiagnosticError("the source-bound adapter requires the V2 model runner") + if os.environ.get("RADIANCE_VERIFY_HEAD", "0") != "0": + raise DiagnosticError( + "use full-vocabulary target logits for the initial conformance replay" + ) + import importlib.util + + from qwen_r9700_lab.conformance_model import Checkpoint + from qwen_r9700_lab.conformance_radiance import verify_sources + + spec = importlib.util.find_spec("vllm") + if spec is None or spec.origin is None: + raise DiagnosticError("the pinned vLLM package is not installed") + verify_sources(Path(spec.origin).parent.parent, private_json(binding_path)) + checkpoint = Checkpoint(Path(plan["checkpoint"]), plan["checkpoint_files"]) + checkpoint.close() + # Everything above runs before importing a GPU library or constructing LLM. + from vllm import LLM, SamplingParams + + extension = "qwen_r9700_lab.conformance_radiance.ConformanceWorkerExtension" + if kwargs.get("worker_extension_cls") not in (None, "", extension): + raise DiagnosticError("native replay requires the qualified worker extension") + kwargs["worker_extension_cls"] = extension + kwargs["disable_log_stats"] = True + llm = LLM(**kwargs) + engine = llm.llm_engine + params = SamplingParams( + temperature=0, + top_p=1, + top_k=-1, + ignore_eos=True, + # The engine process can run ahead of the output consumer. Bound it at + # the final forced token; a client-side abort arrives too late. + max_tokens=len(plan["forced_tokens"]), + detokenize=False, + ) + req = "qwen-isolated-conformance" + try: + with reject_dead_native_rpcs(engine.engine_core): + llm.collective_rpc( + "qwen_conformance_install", + args=(str(plan_path), str(root / "capture"), str(binding_path)), + ) + engine.add_request(req, {"prompt_token_ids": plan["prefix"]}, params) + reached = False + while engine.has_unfinished_requests(): + for output in engine.step(): + if output.request_id != req: + raise DiagnosticError("unexpected request in isolated replay") + if output.outputs and len(output.outputs[0].token_ids) >= len( + plan["forced_tokens"] + ): + reached = True + if reached: + break + if not reached: + raise DiagnosticError("native backend ended before the forced replay completed") + receipt = llm.collective_rpc("qwen_conformance_finish") + write_private( + root / "receipt.json", {"workers": receipt, "native_equivalence": "UNPROVED"} + ) + except Exception as error: + failure_path = root / "capture/worker-error.json" + if failure_path.exists(): + failure = private_json(failure_path) + authenticate(failure) + if ( + failure.get("schema") != "urn:qwen:native-worker-failure:v1" + or failure.get("plan") != plan["sha256"] + or failure.get("binding") != private_json(binding_path)["sha256"] + ): + raise DiagnosticError("native failure receipt belongs to another replay") from error + if failure.get("type") == "DiagnosticError": + raise DiagnosticError(failure["message"]) from error + raise + finally: + engine.abort_request([req]) + + +def run_native(plan, config, binding, root, *, allow_gpu=False): + if not allow_gpu: + raise DiagnosticError( + "GPU use was not authorized; pass --allow-gpu only when the GPU is available" + ) + validate_plan(private_json(plan)) + root.mkdir(mode=0o700) + # The public source binding contains no private data; copy it into the private + # worker input directory so all worker inputs use one strict file policy. + bound = json.loads(binding.read_text()) + authenticate(bound) + write_private(root / "binding.json", bound) + for path, name in ((plan, "plan.json"), (config, "native-config.json")): + write_private(root / name, private_json(path)) + args = [ + sys.executable, + "-m", + "qwen_r9700_lab.conformance_cli", + "_native-worker", + "--output", + str(root.resolve()), + ] + fd = os.open(root / "worker.log", os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(fd, "wb") as log: + env = {**os.environ, "QWEN_CONFORMANCE_GPU": "1"} + result = subprocess.run(args, stdout=log, stderr=log, env=env, check=False) + if result.returncode: + raise DiagnosticError( + "native replay failed; evidence is retained in the private worker log" + ) + return {"captured": True, "native_equivalence": "UNPROVED"} + + +def gate(authority: Path, transition: dict, *, create=False): + required = { + "contract", + "coverage", + "reference", + "candidate", + "base_revision", + "reference_tokens", + "candidate_tokens", + "reference_stop", + "candidate_stop", + } + if set(transition) != required: + raise DiagnosticError("incomplete publication transition") + session = CheckedSession( + authority, + contract=transition["contract"], + required_components=transition["coverage"], + create=create, + ) + try: + receipt = session.commit( + Path(transition["reference"]), + Path(transition["candidate"]), + base_revision=transition["base_revision"], + reference_tokens=tuple(transition["reference_tokens"]), + candidate_tokens=tuple(transition["candidate_tokens"]), + reference_stop=transition["reference_stop"], + candidate_stop=transition["candidate_stop"], + ) + # Raw tokens are available to an authorized consumer by revision, not + # printed to terminal logs by this diagnostic command. + return { + "revision": receipt["revision"], + "output_count": len(receipt["tokens"]), + "receipt": receipt["receipt"], + } + finally: + session.close() + + +def main(argv=None): + parser = argparse.ArgumentParser( + description=HELP, formatter_class=argparse.RawDescriptionHelpFormatter + ) + subs = parser.add_subparsers(dest="command", required=True) + subs.add_parser("inventory") + p = subs.add_parser("proof-obligations") + p.add_argument("--profile", choices=tuple(REFERENCE_PROFILES), default="weight-only-bf16") + p.add_argument("--output", type=Path) + p.add_argument("--require-proved", action="store_true") + p = subs.add_parser("make-plan") + p.add_argument("--spec", type=Path, required=True) + p.add_argument("--output", type=Path, required=True) + p = subs.add_parser("suite-plan") + p.add_argument("--spec", type=Path, required=True) + p.add_argument("--output", type=Path, required=True) + p = subs.add_parser("suite-run") + p.add_argument("--campaign", type=Path, required=True) + p.add_argument("--output", type=Path, required=True) + p.add_argument("--allow-gpu", action="store_true") + p.add_argument("--case", action="append") + p.add_argument("--keep-going", action="store_true") + p.add_argument("--resume", action="store_true") + p.add_argument("--through", choices=("pilot", "focused", "extended"), default="extended") + p.add_argument("--budget-seconds", type=int) + p.add_argument("--retry-failed", action="store_true") + p.add_argument("--min-free-gib", type=int, default=20) + p = subs.add_parser("suite-pause") + p.add_argument("--output", type=Path, required=True) + p = subs.add_parser("suite-status") + p.add_argument("--campaign", type=Path, required=True) + p.add_argument("--results", type=Path, action="append", required=True) + p.add_argument("--require-passing", action="store_true") + for command in ("reference", "native"): + p = subs.add_parser(command) + p.add_argument("--plan", type=Path, required=True) + p.add_argument("--output", type=Path, required=True) + if command == "native": + p.add_argument("--config", type=Path, required=True) + p.add_argument("--binding", type=Path, required=True) + p.add_argument("--allow-gpu", action="store_true") + for command in ("compare", "boundaries", "frame", "calls"): + p = subs.add_parser(command) + p.add_argument("--reference", type=Path, required=True) + p.add_argument("--candidate", type=Path, required=True) + p.add_argument("--output", type=Path, required=True) + p = subs.add_parser("recovery") + for option in ("plan", "reference", "live", "restored", "output"): + p.add_argument("--" + option, type=Path, required=True) + for command, option in (("control", "trace"), ("certificate", "bounds")): + p = subs.add_parser(command) + p.add_argument("--" + option, type=Path, required=True) + p.add_argument("--output", type=Path, required=True) + for command in ("operator", "prove", "_native-worker"): + p = subs.add_parser(command) + p.add_argument("--output", type=Path, required=True) + if command == "operator": + p.add_argument("--capsule", type=Path, required=True) + p = subs.add_parser("gate") + p.add_argument("--authority", type=Path, required=True) + p.add_argument("--transition", type=Path, required=True) + p.add_argument("--create", action="store_true") + args = parser.parse_args(argv) + try: + if args.command == "inventory": + result = inventory() + elif args.command == "proof-obligations": + result = proof_obligations(args.profile) + if args.output is not None: + write_private(args.output, result) + if args.require_proved: + print(json.dumps(result, allow_nan=False)) + return 1 if result["undischarged"] else 0 + elif args.command == "make-plan": + plan, semantics, execution = make_plan(private_json(args.spec)) + write_private(args.output.with_suffix(".semantics.json"), semantics) + write_private(args.output.with_suffix(".execution.json"), execution) + write_private(args.output, plan) + result = {"plan": plan["sha256"], "contract": plan["contract"]} + elif args.command == "suite-pause": + from qwen_r9700_lab.conformance_queue import request_pause + + result = request_pause(args.output) + elif args.command in {"suite-plan", "suite-run", "suite-status"}: + from qwen_r9700_lab.conformance_campaign import ( + build_campaign, + coverage, + load_results, + run_campaign, + ) + + if args.command == "suite-plan": + campaign = build_campaign(private_json(args.spec)) + write_private(args.output, campaign) + result = { + "campaign": campaign["sha256"], + "cases": len(campaign["cases"]), + "gpu_executed": False, + } + else: + campaign = private_json(args.campaign) + if args.command == "suite-run": + report = run_campaign( + campaign, + args.output, + allow_gpu=args.allow_gpu, + selected=args.case, + keep_going=args.keep_going, + resume=args.resume, + through=args.through, + budget_seconds=args.budget_seconds, + retry_failed=args.retry_failed, + min_free_bytes=args.min_free_gib * 1024**3, + ) + else: + report = coverage(campaign, load_results(campaign, args.results)) + result = { + "campaign": report["campaign"], + "counts": report["counts"], + "complete": report["complete"], + "universal_correctness": "UNPROVED", + } + if args.command == "suite-status": + result["checkpoints"] = [] + for directory in args.results: + path = directory / "checkpoint.json" + if path.exists(): + checkpoint = private_json(path) + authenticate(checkpoint) + if checkpoint["campaign"] != campaign["sha256"]: + raise DiagnosticError("checkpoint belongs to another campaign") + result["checkpoints"].append( + { + key: checkpoint.get(key) + for key in ( + "status", + "current", + "through", + "elapsed_this_run", + "updated_ns", + ) + } + ) + print(json.dumps(result, allow_nan=False)) + return ( + 1 + if (args.command == "suite-run" or args.require_passing) + and not report["complete"] + else 0 + ) + elif args.command == "reference": + r = run_reference(private_json(args.plan), args.output) + result = { + "frames": len(r["frames"]), + "schedule": r["sha256"], + "native_equivalence": "UNPROVED", + } + elif args.command == "native": + result = run_native( + args.plan, args.config, args.binding, args.output, allow_gpu=args.allow_gpu + ) + elif args.command == "_native-worker": + root = args.output + try: + native_worker( + root / "plan.json", root / "native-config.json", root / "binding.json", root + ) + except Exception as error: + write_private( + root / "worker-failure.json", + {"type": type(error).__name__, "message": str(error)}, + ) + raise + return 0 + elif args.command in {"compare", "boundaries"}: + fn = compare_campaign if args.command == "compare" else compare_boundaries + r = fn(args.reference, args.candidate, args.output) + result = { + "equal": r["equal"], + "report": r["sha256"], + "first_difference": r["first_difference"], + } + elif args.command == "frame": + result = compare_frames(args.reference, args.candidate) + write_private(args.output, result) + elif args.command == "calls": + result = compare_calls(args.reference, args.candidate, args.output) + elif args.command == "control": + result = audit_control(private_json(args.trace), args.output) + elif args.command == "certificate": + report = interval_certificate(**private_json(args.bounds)) + write_private(args.output, report) + result = { + "equal": report["certified_under_bounds"], + "report": report["sha256"], + "bounds_soundness": report["bounds_soundness"], + } + elif args.command == "recovery": + from qwen_r9700_lab.conformance_model import Checkpoint, state_names + from qwen_r9700_lab.conformance_state import read_frame + + plan = validate_plan(private_json(args.plan)) + checkpoint = Checkpoint(Path(plan["checkpoint"]), plan["checkpoint_files"]) + try: + config = checkpoint.config.get("text_config", checkpoint.config) + if any( + read_frame(p)["contract"] != plan["contract"] + for p in (args.reference, args.live, args.restored) + ): + raise DiagnosticError("recovery frames do not implement the supplied contract") + report = compare_recovery( + args.reference, + args.live, + args.restored, + args.output, + required=state_names(config), + ) + result = { + "equal": report["equal"], + "classification": report["classification"], + "report": report["sha256"], + "native_equivalence": "UNPROVED", + } + finally: + checkpoint.close() + elif args.command == "operator": + result = replay_operator(args.capsule, args.output) + write_private(args.output / "comparison.json", result) + elif args.command == "prove": + result = run_obligations(args.output) + elif args.command == "gate": + result = gate(args.authority, private_json(args.transition), create=args.create) + else: + raise DiagnosticError("unknown conformance action") + print(json.dumps(result, allow_nan=False)) + return ( + 1 if result.get("equal") is False or result.get("all_expected_results") is False else 0 + ) + except SessionMismatchError as error: + print( + json.dumps({"error": str(error), "receipt": error.receipt["sha256"]}), file=sys.stderr + ) + return 1 + except DiagnosticError as error: + print("conformance refused: " + str(error), file=sys.stderr) + return 2 + except (OSError, ValueError, KeyError, TypeError): + # No input data or request text in error logs. Details of a native + # failure remain in its private worker log, not the user's chat. + print( + "conformance refused: invalid/incomplete evidence or unsupported execution; " + "no output published", + file=sys.stderr, + ) + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_control.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_control.py new file mode 100644 index 0000000..3df98c3 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_control.py @@ -0,0 +1,341 @@ +"""Strict event replay for concurrent cache and speculative state lifetimes. + +The event consumer runs on CPU, with no GPU synchronization. A runtime adapter +must emit events at its real boundaries; an event claim is not proof of device +completion or tensor equality. Complete tensor comparisons provide those checks +separately. A truncated trace or unsupported event is never a successful audit. +""" + +from copy import deepcopy +from pathlib import Path + +from qwen_r9700_lab.conformance_invariants import ( + snapshot_publishable, + transaction_ready, + writable_exclusively, +) +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + digest, + integer, + private_json, + require_name, + require_sha, + seal, + validate_speculative_commit, + write_private, +) + +SCHEMA = "urn:qwen:control-events:v1" + + +class ControlLedger: + """Actual audit implementation; inputs contain identities/counts, no text.""" + + def __init__(self): + self.chats, self.transactions, self.blocks, self.snapshots = {}, {}, {}, {} + self.committed, self.restored = 0, 0 + + def apply(self, event): + if not isinstance(event, dict) or "event" not in event: + raise DiagnosticError("malformed control event") + kind = event["event"] + method = getattr(self, "event_" + kind, None) if isinstance(kind, str) else None + if method is None: + raise DiagnosticError("unsupported control event") + # A rejected event cannot partially change the checker state. + saved = deepcopy(self.__dict__) + try: + method(**{k: v for k, v in event.items() if k != "event"}) + except (KeyError, TypeError) as exc: + self.__dict__.update(saved) + raise DiagnosticError("incomplete or unregistered control state") from exc + except Exception: + self.__dict__.update(saved) + raise + + def event_chat(self, chat, generation, consumed, pending, components): + require_name(chat) + require_sha(generation) + integer(consumed) + if type(pending) is not bool or not components or len(set(components)) != len(components): + raise DiagnosticError("incomplete initial control state") + for name in components: + require_name(name) + if chat in self.chats: + raise DiagnosticError("chat registered twice") + self.chats[chat] = { + "generation": generation, + "consumed": consumed, + "pending": int(pending), + "components": set(components), + "revision": 0, + "head": None, + } + + def event_begin(self, transaction, chat, generation, drafted): + require_name(transaction) + state = self.chats[chat] + if transaction in self.transactions or state["generation"] != generation: + raise DiagnosticError("duplicate or stale transition") + if integer(drafted) > 7: + raise DiagnosticError("control profile admits M1 or at most seven drafts") + self.transactions[transaction] = { + "chat": chat, + "generation": generation, + "drafted": drafted, + "revision": state["revision"], + "completed": False, + "validated": False, + "cancelled": False, + "closed": False, + } + + def _active(self, transaction): + tx = self.transactions[transaction] + if tx["closed"]: + raise DiagnosticError("closed transition cannot change state") + return tx + + def event_fence(self, transaction): + tx = self._active(transaction) + if tx["completed"]: + raise DiagnosticError("duplicate completion event") + tx["completed"] = True + + def event_validate(self, transaction, output_equal, state_equal, identity_equal, complete): + tx = self._active(transaction) + values = (output_equal, state_equal, identity_equal, complete) + if any(type(v) is not bool for v in values) or not tx["completed"] or tx["validated"]: + raise DiagnosticError("validation requires completed writes and one comparison") + tx["checks"], tx["validated"] = values, True + + def event_cancel(self, transaction): + tx = self._active(transaction) + tx["cancelled"], tx["closed"] = True, True + + def event_commit(self, transaction, accepted, emitted, consumed, pending, versions): + tx = self._active(transaction) + chat = self.chats[tx["chat"]] + if not tx["validated"] or not transaction_ready( + *tx.get("checks", (False,) * 4), + tx["completed"], + tx["cancelled"], + (tx["generation"], tx["revision"]), + (chat["generation"], chat["revision"]), + ): + raise DiagnosticError("unvalidated, unfinished, cancelled or stale publication") + if type(pending) is not bool: + raise DiagnosticError("pending state must be explicit") + validate_speculative_commit( + before_materialized=chat["consumed"], + before_pending=chat["pending"], + drafted=tx["drafted"], + accepted=accepted, + emitted=emitted, + after_materialized=consumed, + after_pending=int(pending), + component_versions=versions, + required_components=chat["components"], + ) + chat.update(consumed=consumed, pending=int(pending), revision=chat["revision"] + 1) + tx["closed"] = True + self.committed += 1 + + def event_generation(self, chat, generation, consumed, pending): + state = self.chats[chat] + require_sha(generation) + integer(consumed) + if generation == state["generation"] or type(pending) is not bool: + raise DiagnosticError("invalid superseding generation") + state.update( + generation=generation, + consumed=consumed, + pending=int(pending), + revision=state["revision"] + 1, + ) + + def event_allocate(self, block, chat, immutable, external=0): + integer(block) + integer(external) + if chat not in self.chats or block in self.blocks or type(immutable) is not bool: + raise DiagnosticError("invalid allocation or owner") + self.blocks[block] = {"owners": {chat}, "immutable": immutable, "external": external} + + def event_share(self, block, chat): + entry = self.blocks[block] + if chat not in self.chats or chat in entry["owners"] or not entry["immutable"]: + raise DiagnosticError("shared prefix must be immutable with distinct owners") + entry["owners"].add(chat) + + def event_write(self, block, chat): + entry = self.blocks[block] + if entry["owners"] != {chat} or not writable_exclusively( + len(entry["owners"]), entry["external"], entry["immutable"] + ): + raise DiagnosticError("write would modify a shared, pinned or different chat block") + + def event_copy_on_write(self, source, replacement, chat): + if chat not in self.blocks[source]["owners"]: + raise DiagnosticError("copy-on-write source is not owned by chat") + self.event_allocate(replacement, chat, False) + self.event_release(source, chat) + + def event_release(self, block, chat): + entry = self.blocks[block] + if chat not in entry["owners"]: + raise DiagnosticError("release from a different owner") + entry["owners"].remove(chat) + if not entry["owners"] and not entry["external"]: + del self.blocks[block] + + def event_snapshot(self, snapshot, chat, generation, consumed, state_sha256): + require_name(snapshot) + require_sha(state_sha256) + state = self.chats[chat] + if ( + snapshot in self.snapshots + or generation != state["generation"] + or consumed != state["consumed"] + ): + raise DiagnosticError("snapshot belongs to a stale or different state") + self.snapshots[snapshot] = { + "chat": chat, + "generation": generation, + "consumed": consumed, + "state_sha256": state_sha256, + "verified": False, + "durable": False, + } + + def event_verify_snapshot(self, snapshot, state_sha256, durable): + entry = self.snapshots[snapshot] + if state_sha256 != entry["state_sha256"] or type(durable) is not bool: + raise DiagnosticError("snapshot verification changed state or omitted durability") + entry.update(verified=True, durable=durable) + + def event_publish_snapshot(self, snapshot): + entry = self.snapshots[snapshot] + chat = self.chats[entry["chat"]] + if not snapshot_publishable( + entry["verified"], entry["durable"], chat["generation"], entry["generation"] + ): + raise DiagnosticError("snapshot is unverified, nondurable or superseded") + if chat["head"] is not None: + old = self.snapshots[chat["head"]] + if old["generation"] == entry["generation"] and old["consumed"] > entry["consumed"]: + raise DiagnosticError("snapshot publication would rewind the durable head") + chat["head"] = snapshot + + def event_restore(self, snapshot, chat, generation, consumed, state_sha256): + entry, state = self.snapshots[snapshot], self.chats[chat] + if ( + state["head"] != snapshot + or entry["chat"] != chat + or generation != state["generation"] + or generation != entry["generation"] + or consumed != entry["consumed"] + or state_sha256 != entry["state_sha256"] + ): + raise DiagnosticError("restored snapshot identity, version or bytes differ") + self.restored += 1 + + def finish(self): + if not self.chats or not (self.committed or self.restored): + raise DiagnosticError("empty control qualification domain") + if any(not tx["closed"] for tx in self.transactions.values()): + raise DiagnosticError("incomplete trace has unfinished transitions") + return {"commits": self.committed, "restores": self.restored, "chats": len(self.chats)} + + +def audit_control(document: dict, output: Path | None = None): + authenticate(document) + if ( + set(document) != {"schema", "execution", "adapter", "events", "sha256"} + or document["schema"] != SCHEMA + ): + raise DiagnosticError("incomplete control trace identity") + require_sha(document["execution"]) + require_sha(document["adapter"]) + if not isinstance(document["events"], list) or not document["events"]: + raise DiagnosticError("empty control trace") + ledger, failure = ControlLedger(), None + for index, event in enumerate(document["events"]): + try: + ledger.apply(event) + except DiagnosticError as exc: + failure = {"event_index": index, "reason": str(exc)} + break + counts = None + if failure is None: + try: + counts = ledger.finish() + except DiagnosticError as exc: + failure = {"event_index": len(document["events"]), "reason": str(exc)} + report = seal( + { + "schema": "urn:qwen:control-audit:v1", + "trace": document["sha256"], + "equal": failure is None, + "first_failure": failure, + "counts": counts, + "status": "TESTED", + "scope": "complete observed control event sequence", + "device_completion_and_tensor_equality": "ASSUMED: independently check producer events", + "native_adapter_qualification": "UNPROVED", + } + ) + if output is not None: + write_private(output, report) + return report + + +class ControlRecorder: + """Create-once event spool for adapters; survives a crash at any event boundary.""" + + def __init__(self, root: Path, *, execution: str, adapter: str): + import threading + + require_sha(execution) + require_sha(adapter) + root.mkdir(mode=0o700, parents=True, exist_ok=False) + self.root, self.lock, self.count, self.closed = root, threading.Lock(), 0, False + self.identity = {"schema": SCHEMA, "execution": execution, "adapter": adapter} + write_private(root / "identity.json", seal(self.identity)) + self.previous = digest(self.identity) + + def record(self, event: str, **fields): + with self.lock: + if self.closed: + raise DiagnosticError("control recorder already finalized") + entry = seal( + {"index": self.count, "previous": self.previous, "data": {"event": event, **fields}} + ) + write_private(self.root / f"{self.count:09d}.json", entry) + self.previous, self.count = entry["sha256"], self.count + 1 + + def finish(self): + with self.lock: + if self.closed: + raise DiagnosticError("control recorder already finalized") + self.closed = True + doc = read_control_spool(self.root, count=self.count) + write_private(self.root / "events.json", doc) + return audit_control(doc) + + +def read_control_spool(root: Path, *, count: int): + integer(count) + identity = private_json(root / "identity.json") + authenticate(identity) + identity = {k: v for k, v in identity.items() if k != "sha256"} + events, previous = [], digest(identity) + for index in range(count): + entry = private_json(root / f"{index:09d}.json") + authenticate(entry) + if entry["index"] != index or entry["previous"] != previous: + raise DiagnosticError("control event chain is reordered, missing or changed") + events.append(entry["data"]) + previous = entry["sha256"] + return seal({**identity, "events": events}) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_dispatch.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_dispatch.py new file mode 100644 index 0000000..2f8af4b --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_dispatch.py @@ -0,0 +1,184 @@ +"""Opt-in, library-bound entrypoint tracing, without device reads or GPU imports. + +Python aliases of reviewed native exports are replaced too. This observes calls +through those entrypoints, not launches hidden in C++, graphs, or a captured +closure. A library hash is deliberately never called an ISA attestation. +""" + +from __future__ import annotations + +import functools +import threading +import time +from pathlib import Path + +from qwen_r9700_lab.conformance_artifacts import file_identity +from qwen_r9700_lab.conformance_instrumentation import HookSet +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + require_name, + require_sha, + seal, + write_private, +) + + +def replace_aliases(value, replacements, ancestors=frozenset()): + """Copy alias containers; do not mutate a list retained by another owner.""" + replacement = replacements.get(id(value)) + if replacement is not None: + return replacement, True + if isinstance(value, (list, tuple, dict)): + if id(value) in ancestors: + raise DiagnosticError("cyclic native alias container is unsupported") + ancestors = ancestors | {id(value)} + if isinstance(value, (list, tuple)): + children = [replace_aliases(child, replacements, ancestors) for child in value] + if any(changed for _, changed in children): + return type(value)(child for child, _ in children), True + elif isinstance(value, dict): + children = { + key: replace_aliases(child, replacements, ancestors) for key, child in value.items() + } + if any(changed for _, changed in children.values()): + return {key: child for key, (child, _) in children.items()}, True + return value, False + + +class DispatchRecorder: + def __init__(self, root: Path, *, execution: str): + self.execution = require_sha(execution) + root.mkdir(mode=0o700) + self.root, self.hooks = root, HookSet() + self.bindings, self.rows, self.required, self.aliases = {}, [], set(), [] + self.lock, self.closed = threading.Lock(), False + + def bind(self, module, binding, *, aliases=()): + """Bind a reviewed, complete public callable export set before replay. + + binding: sealed {schema, module, library_sha256, exports, required}. + exports maps names to kernel/metadata; required names must execute. + No metadata functions (including registry helpers) are invoked here. + """ + authenticate(binding) + if set(binding) != {"schema", "module", "library_sha256", "exports", "required", "sha256"}: + raise DiagnosticError("incomplete native entrypoint binding") + if binding["schema"] != "urn:qwen:native-entrypoint-binding:v1": + raise DiagnosticError("unsupported native entrypoint binding") + name = require_name(binding["module"]) + if module.__name__ != name or name in self.bindings: + raise DiagnosticError("native entrypoint module identity mismatch") + exports = binding["exports"] + observed = { + key + for key, value in vars(module).items() + if not key.startswith("_") and callable(value) + } + if ( + not exports + or set(exports) != observed + or any(kind not in {"kernel", "metadata"} for kind in exports.values()) + ): + raise DiagnosticError("native export inventory changed or is incomplete") + required = binding["required"] + if ( + not isinstance(required, list) + or not required + or len(set(required)) != len(required) + or any(exports.get(key) != "kernel" for key in required) + ): + raise DiagnosticError("native binding needs an explicit nonempty exercised domain") + if file_identity(Path(module.__file__))["sha256"] != require_sha(binding["library_sha256"]): + raise DiagnosticError("native library changed before entrypoint binding") + replacements = {} + for symbol, kind in exports.items(): + require_name(symbol) + if kind != "kernel": + continue + original = getattr(module, symbol) + if id(original) in replacements: + raise DiagnosticError("native exports alias each other; mapping is ambiguous") + site = name + "." + symbol + + @functools.wraps(original) + def call(*args, _site=site, _fn=original, **kwargs): + with self.lock: + if self.closed: + raise DiagnosticError("native dispatch recorder already finalized") + row = { + "index": len(self.rows), + "site": _site, + "started_ns": time.monotonic_ns(), + "completed": False, + "argument_count": len(args), + "keyword_count": len(kwargs), + } + self.rows.append(row) + path = self.root / f"entry-{row['index']:09d}" + write_private(path.with_suffix(".started.json"), seal(row)) + try: + result = _fn(*args, **kwargs) + row["completed"] = True + return result + except BaseException as exc: + row["exception_type"] = type(exc).__qualname__ + raise + finally: + row["ended_ns"] = time.monotonic_ns() + write_private(path.with_suffix(".finished.json"), seal(row)) + + replacements[id(original)] = call + # Validate every export before replacing any attribute. + changes = [] + for owner in (module, *aliases): + for attribute, value in list(vars(owner).items()): + # Skip arbitrary objects/closures and cyclic runtime internals. + if attribute.startswith("__"): + continue + replacement, changed = replace_aliases(value, replacements) + if changed: + changes.append((owner, attribute, replacement)) + for owner, attribute, replacement in changes: + self.hooks.replace(owner, attribute, replacement) + self.aliases.append({"owner": owner.__name__, "attribute": attribute}) + self.bindings[name] = {"binding": binding, "library": str(module.__file__)} + self.required.update(name + "." + symbol for symbol in required) + + def finish(self): + try: + with self.lock: + self.closed = True + if not self.rows or any(not row["completed"] for row in self.rows): + raise DiagnosticError("native entrypoint observation is incomplete") + if not self.required <= {row["site"] for row in self.rows}: + raise DiagnosticError("required native entrypoint was not exercised") + for entry in self.bindings.values(): + if ( + file_identity(Path(entry["library"]))["sha256"] + != entry["binding"]["library_sha256"] + ): + raise DiagnosticError("native library changed during capture") + result = seal( + { + "schema": "urn:qwen:native-entrypoint-capture:v1", + "execution": self.execution, + "bindings": self.bindings, + "aliases": self.aliases, + "calls": self.rows, + "scope": "observed library entrypoints and explicit Python aliases only", + "device_completion": "UNPROVED; return does not mean device completion", + "argument_and_state_equivalence": ( + "UNPROVED; pointer values are not tensor evidence" + ), + "hidden_dispatch_and_graphs": "UNPROVED", + "exact_device_binary_attested": False, + } + ) + write_private(self.root / "dispatch.json", result) + return result + finally: + self.hooks.close() + + def close(self): + self.hooks.close() diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_execution_modes.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_execution_modes.py new file mode 100644 index 0000000..d8c3209 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_execution_modes.py @@ -0,0 +1,151 @@ +"""Admit a controlled execution-mode comparison, never a universal correctness claim.""" + +from qwen_r9700_lab.conformance_topk import aggregate, compare_saved_rows, require +from qwen_r9700_lab.diagnostic_contract import authenticate, digest, seal + +MODES = {"compiled", "compiled-no-graphs", "eager"} +CAPACITY = { + "max_num_seqs", + "max_num_batched_tokens", + "max_model_len", + "block_size", + "cache_dtype", + "num_gpu_blocks", +} +SOURCES = { + "optimized_d7_worker.py", + "optimized_stock_norm.py", + "execution_mode_d7_worker.py", + "isolated_d7_capture.py", +} + + +def numerical_config(config): + # Everything except these explicitly varied execution controls must match. + # In particular keep request capacity, prefill chunk size, KV, seed, model, + # speculation and sampling defaults. Do not blacklist unknown future fields. + result = { + k: v + for k, v in config.items() + if k + not in { + "sha256", + "enforce_eager", + "compilation_config", + "worker_cls", + } + } + result["compilation_config"] = { + k: v + for k, v in config.get("compilation_config", {}).items() + if k not in {"mode", "cudagraph_mode", "cudagraph_capture_sizes"} + } + return result + + +def admit_pair(left, right, *, arm="m8"): + """Each side supplies authenticated measurement, requested config and runtime.""" + require(arm in {"m1", "m8"}, "unknown execution-mode comparison arm") + for side in (left, right): + for key in ("measurement", "config", "runtime", "pass"): + authenticate(side[key]) + m, c, r = (side[k] for k in ("measurement", "config", "runtime")) + mode = m.get("execution_mode") + require(mode in MODES, "missing or invalid execution-mode identity") + require( + m["correctness"] and m["m1"] == (arm == "m1"), + f"comparison requires forced {arm.upper()} replay on both sides", + ) + require(m["lanes"] == ["fixed-bf16"], "comparison requires the same final fixed lane") + eager = mode == "eager" + require(c["enforce_eager"] == r["enforce_eager"] == eager, "eager mode was not honored") + require((r["compilation_mode"] == 0) == eager, "unexpected compiler mode") + graph = "NONE" if mode != "compiled" or m["isolated_capture"] else "PIECEWISE" + require(str(r["graph_mode"]).split(".")[-1] == graph, "unexpected graph execution") + require(r["async_scheduling"] is False, "comparison requires ordered scheduling") + p = side["pass"] + require( + p["mode"] == "correctness" + and p["execution_mode"] == mode + and p["fixture"] == m["fixture"], + "pass receipt does not match the requested execution", + ) + counts = p["observation"]["observation"]["counts"] + require(counts.get("target_forward_calls", 0) > 0, "no target execution observed") + require( + (counts.get("target_graph_replays", 0) > 0) == (graph == "PIECEWISE"), + "observed graph execution contradicts configuration", + ) + require(set(r.get("effective_capacity", {})) == CAPACITY, "actual capacity not recorded") + require(all(v is not None for v in r["effective_capacity"].values()), "unknown capacity") + require(set(r.get("diagnostic_sources", {})) == SOURCES, "unbound integration sources") + require((r.get("repair") or {}).get("bundle"), "missing repair identity") + require((r.get("performance") or {}).get("manifest"), "missing performance identity") + authenticate(r["runtime"]) + for key in ("python", "kernel", "machine", "packages", "flags"): + require(key in r["runtime"], f"runtime identity lacks {key}") + + lm, lc, lr = (left[k] for k in ("measurement", "config", "runtime")) + rm, rc, rr = (right[k] for k in ("measurement", "config", "runtime")) + for key in ("fixture", "binding", "driver_sha256", "prefix_tokens"): + require(lm[key] == rm[key], f"comparison changes {key}") + require(numerical_config(lc) == numerical_config(rc), "non-execution configuration differs") + for key in ("effective_capacity", "diagnostic_sources"): + require(lr[key] == rr[key], f"comparison changes {key}") + for key in ("python", "kernel", "machine", "packages", "flags"): + require(lr["runtime"][key] == rr["runtime"][key], f"runtime changes {key}") + require( + lr["runtime"].get("compiler_settings") == rr["runtime"].get("compiler_settings"), + "runtime changes observed compiler settings", + ) + require(lr["repair"]["bundle"] == rr["repair"]["bundle"], "repair bundles differ") + require(lr["performance"]["manifest"] == rr["performance"]["manifest"], "performance differs") + return seal( + { + "schema": "qwen.execution-mode-admission.v1", + "status": "ADMITTED_MODE_COMPARISON", + "modes": [lm["execution_mode"], rm["execution_mode"]], + "captures": [lm["isolated_capture"], rm["isolated_capture"]], + "arm": arm, + "fixture": lm["fixture"], + "binding": lm["binding"], + "numerical_config_sha256": digest(numerical_config(lc)), + "capacity": lr["effective_capacity"], + "sources": lr["diagnostic_sources"], + "repair": lr["repair"]["bundle"], + "performance": lr["performance"]["manifest"], + "observed_compiler_settings": lr["runtime"].get("compiler_settings"), + "receipts": [ + [s[k]["sha256"] for k in ("measurement", "config", "runtime", "pass")] + for s in (left, right) + ], + "scope": "controlled replay admission; mode-selected operators may differ", + } + ) + + +def compare_pair(left, right, left_rows, right_rows, *, arm="m8"): + admission = admit_pair(left, right, arm=arm) + for side, rows in zip((left, right), (left_rows, right_rows), strict=True): + authenticate(rows) + require( + side["pass"]["observation"]["forced"]["sha256"] == rows["sha256"], + "saved rows do not belong to this observed pass", + ) + require(rows["continuation"] == admission["fixture"], "rows belong to another fixture") + require(len(rows["rows"]) == 320, "mode comparison requires all 320 decode positions") + decoded, initial = compare_saved_rows( + left_rows, right_rows, target_rows=1 if arm == "m1" else 8 + ) + return seal( + { + "schema": "qwen.execution-mode-comparison.v1", + "status": "COMPARED_SAVED_EVIDENCE", + "admission": admission, + "row_sources": [left_rows["sha256"], right_rows["sha256"]], + "decode": aggregate(decoded), + "prefill": aggregate([initial]), + "stage_localization": "UNMEASURED; inspect aligned boundary captures separately", + "scope": "320 forced decode predictions and one prefill prediction; no universal proof", + } + ) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_faults.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_faults.py new file mode 100644 index 0000000..a6c6fe2 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_faults.py @@ -0,0 +1,128 @@ +"""Explicit faults in owned native replay workers; never imported by production. + +Faults alter the actual selected device allocation before the normal extractor +reads it. A receipt records application separately from detection. A failed +startup can therefore never count as a successful corruption negative control. +""" + +from __future__ import annotations + +import os +import re +from pathlib import Path + +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, private_json, write_private + +FAULTS = ("kv", "gdn", "conv", "pending", "position", "version", "missing_observation") + + +def install_experiment(probe): + path = os.environ.get("QWEN_CONFORMANCE_NATIVE_EXPERIMENT") + if not path: + return + if os.environ.get("QWEN_CONFORMANCE_GPU") != "1": + raise DiagnosticError("native fault experiment is not armed") + spec = private_json(Path(path)) + if set(spec) - {"fault", "step", "rejected_suffix_token"}: + raise DiagnosticError("unsupported native fault configuration") + fault, step = spec.get("fault"), spec.get("step", 0) + if fault is not None and fault not in FAULTS: + raise DiagnosticError("unknown native corruption control") + if type(step) is not int or step < 0 or step >= len(probe.campaign.expected): + raise DiagnosticError("fault target is outside the replay schedule") + suffix = spec.get("rejected_suffix_token") + if suffix is not None: + if type(suffix) is not int or not 0 <= suffix < probe.config["vocab_size"]: + raise DiagnosticError("invalid rejected-suffix token") + probe.rejected_suffix_token = suffix + if fault is None: + return + original = probe.capture + applied = False + + def capture(batch, expected, logits, **kwargs): + nonlocal applied + if not applied and probe.index == step: + component = mutate_device(probe, batch, fault) + applied = True + write_private( + probe.campaign.root / "fault-applied.json", + { + "fault": fault, + "step": step, + "component": component, + "consumed": expected["consumed"], + "actual_device_allocation": fault != "missing_observation", + }, + ) + if fault == "missing_observation": + return + return original(batch, expected, logits, **kwargs) + + probe.hooks.replace(probe, "capture", capture) + + +def mutate_device(probe, batch, fault): + if os.environ.get("QWEN_CONFORMANCE_GPU") != "1": + raise DiagnosticError("native device mutation is not armed") + import torch + from vllm.model_executor.layers.mamba.mamba_utils import is_conv_state_dim_first + + from qwen_r9700_lab.conformance_radiance import convolution_offset, temporal_column + + torch.cuda.synchronize() + runner, c = probe.runner, probe.config + index = int(batch.idx_mapping_np[0]) + if fault == "missing_observation": + return "observation.frame" + if fault == "position": + runner.req_states.num_computed_tokens.gpu[index] += 1 + return "sequence.position" + if fault == "pending": + value = runner.req_states.last_sampled_tokens[index, 0] + value.copy_((value + 1) % c["vocab_size"]) + return "sequence.pending" + if fault == "version": + # Invalid version, rather than a version that happens to contain equal data. + runner.model_state._mamba_state_idx_gpu[index] = -1 + return "gdn.version" + targets = {id(m) for _, m in runner.model.named_modules()} + context = runner.compilation_config.static_forward_context + for gid, group in enumerate(runner.kv_cache_config.kv_cache_groups): + table = runner.block_tables.block_tables[gid].gpu[index] + for name in group.layer_names: + module = context[name] + if id(module) not in targets: + continue + match = re.search(r"(?:^|\.)layers\.(\d+)\.", name) + if match is None: + raise DiagnosticError("unmapped fault target") + layer = int(match.group(1)) + linear = c["layer_types"][layer] == "linear_attention" + if fault == "kv" and not linear: + # First logically consumed key byte. Do not reshape a + # noncontiguous view: reshape could mutate a disposable copy. + selected = module.kv_cache[int(table[0]), 0, 0, 0:1] + elif fault in {"gdn", "conv"} and linear: + running = int(runner.model_state._mamba_state_idx_gpu[index]) + accepted = int(runner.model_state.num_accepted_tokens_gpu[index]) + conv, state = module.kv_cache + if fault == "gdn": + selected = state[int(table[temporal_column(running, accepted)])] + selected = selected[(0,) * (selected.ndim - 1) + (slice(0, 1),)] + else: + offset = convolution_offset(accepted) + selected = conv[int(table[running])] + selected = ( + selected[0, offset : offset + 1] + if is_conv_state_dim_first() + else selected[offset, 0:1] + ) + else: + continue + if selected.numel() != 1 or not selected.is_contiguous(): + raise DiagnosticError("native fault target is not an actual scalar view") + selected.view(torch.uint8)[0].bitwise_xor_(1) + torch.cuda.synchronize() + return f"layer.{layer:03d}.{'keys' if fault == 'kv' else fault}" + raise DiagnosticError("required native corruption target was not found") diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_gate.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_gate.py new file mode 100644 index 0000000..1c235be --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_gate.py @@ -0,0 +1,170 @@ +"""Experimental exact output/state publication gate. + +This is an in-process authority for isolated conformance runs, not a production +Pi proxy. Native workers must be independently isolated and supply complete +canonical state bytes. Their isolation, reference arithmetic and serialization +are separate obligations; this gate cannot infer them from matching hashes. +""" + +from __future__ import annotations + +import threading +from collections.abc import Mapping +from dataclasses import dataclass +from types import MappingProxyType +from typing import Any + +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + digest, + integer, + require_name, + require_sha, + seal, +) + + +def publication_allowed(output_equal, state_equal, identity_equal, complete): + """Pure Boolean predicate also executed symbolically by the proof runner.""" + return output_equal & state_equal & identity_equal & complete + + +def copy_accepted_prefix(current: tuple, proposed: tuple, count: int) -> tuple: + """Small actual reference helper; never keep the rejected proposed suffix.""" + if type(count) is not int or len(current) != len(proposed) or not 0 <= count <= len(current): + raise DiagnosticError("invalid prefix-copy domain") + return proposed[:count] + current[count:] + + +@dataclass(frozen=True) +class TentativeTransition: + base_revision: int + semantics_sha256: str + input_sha256: str + consumed_tokens: int + pending_token: bytes | None + state: Mapping[str, bytes] + output_events: tuple[bytes, ...] + stop_reason: str | None + + def __post_init__(self): + integer(self.base_revision) + integer(self.consumed_tokens) + require_sha(self.semantics_sha256) + require_sha(self.input_sha256) + if self.pending_token is not None and type(self.pending_token) is not bytes: + raise DiagnosticError("pending token must be immutable canonical bytes") + if self.stop_reason not in {None, "eos", "tool_call", "complete"}: + raise DiagnosticError("errors and truncation are not successful stop events") + components = {} + for name, payload in self.state.items(): + require_name(name) + if type(payload) is not bytes: + raise DiagnosticError("state components must be immutable canonical bytes") + components[name] = payload + if type(self.output_events) is not tuple or any( + type(v) is not bytes for v in self.output_events + ): + raise DiagnosticError("output events must be immutable canonical bytes") + object.__setattr__(self, "state", MappingProxyType(components)) + + +class ConformanceMismatchError(DiagnosticError): + def __init__(self, report: dict[str, Any]): + super().__init__("candidate did not match the reference; no transition published") + self.report = report + + +class CheckedAuthority: + """Commit state/events together after exact comparison and revision checking. + + No callback receives candidate events while they are tentative. Publication + is the return value after the authority's revision is advanced. Durability, + an external Pi delivery channel and worker isolation are not implemented here. + """ + + def __init__(self, semantics_sha256: str, required_components: set[str], *, sampler="greedy"): + self.semantics_sha256 = require_sha(semantics_sha256) + if sampler != "greedy": + raise DiagnosticError("sampled execution needs a separate distribution contract") + if not required_components: + raise DiagnosticError("state gate cannot admit empty component coverage") + self.required_components = frozenset(require_name(v) for v in required_components) + self._lock = threading.Lock() + self._revision = 0 + self._transition = None + + @property + def revision(self): + with self._lock: + return self._revision + + def compare_and_commit( + self, + reference: TentativeTransition, + candidate: TentativeTransition, + ) -> tuple[bytes, ...]: + with self._lock: + identity_equal = bool( + reference.base_revision == candidate.base_revision == self._revision + and reference.semantics_sha256 + == candidate.semantics_sha256 + == self.semantics_sha256 + and reference.input_sha256 == candidate.input_sha256 + ) + complete = bool( + set(reference.state) == set(candidate.state) == self.required_components + ) + # Compare actual bytes, not digests. A hash collision cannot admit + # an incorrect transition through this gate. + output_equal = bool( + reference.output_events == candidate.output_events + and reference.stop_reason == candidate.stop_reason + ) + state_equal = bool( + reference.state == candidate.state + and reference.consumed_tokens == candidate.consumed_tokens + and reference.pending_token == candidate.pending_token + ) + if not publication_allowed(output_equal, state_equal, identity_equal, complete): + differing = [ + name + for name in sorted(self.required_components) + if reference.state.get(name) != candidate.state.get(name) + ] + report = seal( + { + "schema": "urn:qwen:conformance-gate-mismatch:v1", + "revision": self._revision, + "identity_equal": identity_equal, + "complete_state_coverage": complete, + "output_equal": output_equal, + "state_equal": state_equal, + "first_different_component": differing[0] if differing else None, + "reference_input_sha256": reference.input_sha256, + "candidate_input_sha256": candidate.input_sha256, + "published": False, + "scope": "isolated_in_process_authority", + } + ) + raise ConformanceMismatchError(report) + # Retain the trusted reference state; the candidate cannot substitute + # a later mutable buffer after its bytes have been checked. + self._transition = reference + self._revision += 1 + return candidate.output_events + + def evidence(self) -> dict[str, Any]: + with self._lock: + return { + "revision": self._revision, + "semantics_sha256": self.semantics_sha256, + "components_sha256": digest(sorted(self.required_components)), + "publication_gate": "TESTED", + "native_reference_adapter": "UNPROVED", + "native_state_completeness": "UNPROVED", + "worker_isolation": "ASSUMED", + "python_runtime_and_hardware": "ASSUMED", + "production_pi_integration": "UNPROVED", + "formal_backend_equivalence": "UNPROVED", + } diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_gdn_contract.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_gdn_contract.py new file mode 100644 index 0000000..30454c2 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_gdn_contract.py @@ -0,0 +1,116 @@ +"""Pinned stock-M1 arithmetic identity for future GDN repair comparisons. + +This CPU-only audit checks retained sources and compiler artifacts. It does not +load a GPU library, attest a dispatched kernel, change a replay plan or qualify a +candidate. Existing NumPy and R4D references keep their own semantic identities. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +from pathlib import Path + +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + integer, + seal, + write_private, +) + +CONTRACT_ID = "stock-packed-gdn-m1-gfx1201-v1" +CONTRACT_SHA256 = "d8149b00210c94acfca4f1f66030de302dcaf8155894de6a23d3f4b76da61e00" + + +def read_contract(path: Path) -> dict: + result = json.loads(path.read_text()) + authenticate(result) + if result.get("id") != CONTRACT_ID or result["sha256"] != CONTRACT_SHA256: + raise DiagnosticError("unreviewed GDN arithmetic contract; use a distinct revision") + return result + + +def _check_file(path: Path, expected: str) -> dict: + with path.open("rb") as source: + actual = hashlib.file_digest(source, "sha256").hexdigest() + if actual != expected: + raise DiagnosticError(f"GDN arithmetic evidence changed: {path.name}") + return {"sha256": actual, "bytes": path.stat().st_size} + + +def audit_artifacts(contract: dict, compiled_root: Path, sources: dict[str, Path]) -> dict: + authenticate(contract) + if contract.get("id") != CONTRACT_ID or contract["sha256"] != CONTRACT_SHA256: + raise DiagnosticError("unreviewed GDN arithmetic contract; use a distinct revision") + authority = contract["arithmetic_authority"] + if set(sources) != set(authority["source"]): + raise DiagnosticError("incomplete GDN source evidence") + checked_sources = { + name: _check_file(sources[name], expected) for name, expected in authority["source"].items() + } + checked_artifacts = { + name: _check_file(compiled_root / name, expected) + for name, expected in authority["files"].items() + } + # The full metadata file is already hash-bound; these explicit comparisons + # also keep the human-readable contract consistent with the retained build. + metadata = json.loads((compiled_root / (authority["kernel"] + ".json")).read_text()) + if metadata["target"] != authority["target"] or any( + metadata.get(key) != value for key, value in authority["compiler"].items() + ): + raise DiagnosticError("GDN compiler metadata contradicts the declared arithmetic") + return seal( + { + "schema": "urn:qwen:gdn-arithmetic-artifact-audit:v1", + "contract": contract["sha256"], + "auditor_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(), + "sources": checked_sources, + "artifacts": checked_artifacts, + "artifact_identity": "TESTED", + "runtime_dispatch": "UNPROVED", + "candidate_equivalence": "UNPROVED", + "full_model_equivalence": "UNPROVED", + "gpu_used": False, + "installed_backend_changed": False, + "existing_matrix_contract_changed": False, + } + ) + + +def committed_gdn_row(accepted_proposals: int) -> int: + """Index into eight after-row states, with row zero consuming pending input. + + The correction/bonus token emitted by this round is still pending. This is + an index into *after-row* states, not a count or an initial-inclusive frame. + """ + integer(accepted_proposals) + if accepted_proposals > 7: + raise DiagnosticError("D7 admits zero through seven accepted proposals") + return accepted_proposals + + +def required_processed_rows(accepted_proposals: int) -> tuple[int, ...]: + return tuple(range(committed_gdn_row(accepted_proposals) + 1)) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--contract", type=Path, required=True) + parser.add_argument("--compiled-root", type=Path, required=True) + parser.add_argument("--stock-source", type=Path, required=True) + parser.add_argument("--op-source", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + report = audit_artifacts( + read_contract(args.contract), + args.compiled_root, + {"fused_recurrent.py": args.stock_source, "fla_op.py": args.op_source}, + ) + write_private(args.output, report) + print(json.dumps(report, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_gpu_lease.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_gpu_lease.py new file mode 100644 index 0000000..af43b6e --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_gpu_lease.py @@ -0,0 +1,130 @@ +"""Cooperative GPU ownership for concurrent qualification controllers. + +The optional lock serializes GPU phases, allowing independent CPU reference +preparation to overlap. It does not claim to exclude unrelated GPU programs. +""" + +from __future__ import annotations + +import contextlib +import fcntl +import os +import stat +import time +from pathlib import Path + +from qwen_r9700_lab.conformance_queue import replace_private +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + private_json, + seal, + write_private, +) + +_FAILED_CLEANUP = {} + + +def cleanup_block(): + """A released flock is insufficient when an owned worker could not exit.""" + name = os.environ.get("QWEN_CONFORMANCE_GPU_LOCK") + if name is None: + return None + path = Path(name) + if not name or not path.is_absolute(): + raise DiagnosticError("qualification GPU lock must be an explicit absolute path") + if name in _FAILED_CLEANUP: + return _FAILED_CLEANUP[name] + marker = path.with_name(path.name + ".blocked.json") + if not marker.exists(): + return None + receipt = private_json(marker) + authenticate(receipt) + return receipt + + +def block_cleanup(evidence, *, process_group, reason, members): + receipt = seal( + { + "status": "cleanup_incomplete", + "evidence": str(evidence), + "process_group": process_group, + "reason": reason, + "members": members, + "observed_ns": time.time_ns(), + } + ) + name = os.environ.get("QWEN_CONFORMANCE_GPU_LOCK") + if name is not None: + # Do not report a clean release if writing the diagnostic itself fails. + _FAILED_CLEANUP[name] = receipt + path = Path(name) + if not path.is_absolute(): + raise DiagnosticError("qualification GPU lock must be absolute") + replace_private(path.parent, path.name + ".blocked.json", receipt) + write_private(Path(evidence) / "cleanup-incomplete.json", receipt) + + +@contextlib.contextmanager +def gpu_lease(evidence: Path): + name = os.environ.get("QWEN_CONFORMANCE_GPU_LOCK") + if name is None: + yield + return + path = Path(name) + if not name or not path.is_absolute(): + raise DiagnosticError("qualification GPU lock must be an explicit absolute path") + evidence.mkdir(mode=0o700) + requested = time.monotonic() + write_private( + evidence / "requested.json", + seal({"pid": os.getpid(), "lock": str(path), "requested_ns": time.time_ns()}), + ) + descriptor = os.open(path, os.O_RDWR | os.O_CREAT | os.O_NOFOLLOW | os.O_CLOEXEC, 0o600) + try: + if not stat.S_ISREG(os.fstat(descriptor).st_mode): + raise DiagnosticError("qualification GPU lock must be a regular file") + fcntl.flock(descriptor, fcntl.LOCK_EX) + if cleanup_block() is not None: + raise DiagnosticError( + "qualification worker cleanup is incomplete; GPU admission blocked" + ) + owner_name = path.name + ".owner.json" + owner_path = path.with_name(owner_name) + if owner_path.exists(): + previous = private_json(owner_path) + authenticate(previous) + if previous.get("status") != "released": + raise DiagnosticError( + "previous qualification GPU owner did not release cleanly; " + "verify its process group has exited before clearing the lease" + ) + owner = {"pid": os.getpid(), "evidence": str(evidence), "status": "active"} + replace_private(path.parent, owner_name, seal(owner)) + write_private( + evidence / "acquired.json", + seal({"pid": os.getpid(), "wait_seconds": time.monotonic() - requested}), + ) + try: + yield + finally: + blocked = cleanup_block() is not None + write_private( + evidence / ("scope-cleanup-incomplete.json" if blocked else "released.json"), + seal( + { + "pid": os.getpid(), + "blocked_ns" if blocked else "released_ns": time.time_ns(), + } + ), + ) + replace_private( + path.parent, + owner_name, + seal({**owner, "status": "cleanup_incomplete" if blocked else "released"}), + ) + finally: + # A killed worker leaves an active owner receipt. The next controller + # fails closed even if the kernel lock has gone away, since GPU child + # processes may still require cleanup. + os.close(descriptor) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_instrumentation.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_instrumentation.py new file mode 100644 index 0000000..b2ecbf4 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_instrumentation.py @@ -0,0 +1,392 @@ +"""Portable, source-bound call instrumentation and exact tensor capture. + +No backend is imported here. Native adapters supply tensor export and logical +position mapping explicitly. Tensor mode synchronizes through that exporter; +metadata mode never exports a tensor and cannot claim numerical equivalence. +""" + +from __future__ import annotations + +import contextvars +import functools +import hashlib +import inspect +import marshal +import threading +import time +from pathlib import Path + +import numpy as np + +from qwen_r9700_lab.conformance_state import FrameWriter, compare_frames, read_frame +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + digest, + private_json, + require_name, + require_sha, + seal, + write_private, +) + +CALL_SCHEMA = "urn:qwen:semantic-calls:v1" + + +def _is_custom_op(function): + cls = type(function) + return cls.__module__ == "torch._library.custom_ops" and cls.__qualname__ == "CustomOpDef" + + +def _custom_op_registration(function): + items = function._backend_fns.items() + if any(key is not None and not isinstance(key, str) for key in function._backend_fns): + raise DiagnosticError("unsupported custom operator backend registry") + return ( + function._init_fn, + tuple(sorted(items, key=lambda item: (item[0] is not None, item[0] or ""))), + frozenset(function._disabled_kernel), + function._schema, + ) + + +def callable_identity(function): + if _is_custom_op(function): + initial, backends, disabled, schema = _custom_op_registration(function) + registered = [] + for device, wrapper in backends: + # CustomOpDef wraps each registered implementation in a dispatcher + # closure. Binding just __call__ or the default body would miss a + # different CUDA implementation under the same operator name. + implementation = inspect.getclosurevars(inspect.unwrap(wrapper)).nonlocals.get("fn") + if implementation is None: + raise DiagnosticError("custom operator has an uninspectable registered kernel") + registered.append( + { + "device": device, + "wrapper": callable_identity(wrapper), + "implementation": callable_identity(implementation), + } + ) + return digest( + { + "kind": "torch.library.custom_op", + "name": function._qualname, + "schema": schema, + "dispatcher": callable_identity(type(function).__call__), + "initial": callable_identity(initial), + "registered": registered, + "disabled": sorted(disabled), + } + ) + target = inspect.unwrap(function) + try: + source = inspect.getsource(target).encode() + except (TypeError, OSError) as exc: + raise DiagnosticError("call site requires an explicit inspectable source binding") from exc + dispatched = getattr(function, "__func__", function) + code = getattr(dispatched, "__code__", None) + if code is None: + raise DiagnosticError("call site requires an explicit executable code binding") + # getsource/unwrap alone would name the original function even if a wrapper + # with different executable bytecode actually receives the invocation. + return digest( + { + "source": hashlib.sha256(source).hexdigest(), + "python_code": hashlib.sha256(marshal.dumps(code)).hexdigest(), + } + ) + + +class HookSet: + """Restore the exact previous attribute, including class-bound methods. + + This is diagnostic instrumentation, not a process-wide singleton. An alias + cached elsewhere is a separate binding and cannot be claimed as observed. + """ + + def __init__(self): + self.entries = [] + + def replace(self, owner, name, replacement): + if any(obj is owner and key == name for obj, key, *_ in self.entries): + raise DiagnosticError("call site already instrumented") + original = getattr(owner, name) + own = name in vars(owner) + setattr(owner, name, replacement) + self.entries.append((owner, name, original, own, replacement)) + + def close(self): + changed = [] + for owner, name, original, own, replacement in reversed(self.entries): + if getattr(owner, name, None) is not replacement: + changed.append(name) + continue # do not overwrite somebody else's later instrumentation + if own: + setattr(owner, name, original) + else: + delattr(owner, name) + self.entries.clear() + if changed: + raise DiagnosticError("instrumented call site changed before cleanup") + + +class CallRecorder: + """Capture immutable inputs *before* a call, mutated arguments and outputs after. + + Bindings identify actual Python functions called, not the selected HSACO. + The latter needs a device-dispatch receipt. Unknown argument objects are + explicitly inventoried, not incorrectly labelled complete tensor state. + """ + + def __init__( + self, + root: Path, + *, + contract, + execution, + adapter, + tensor_export, + is_tensor, + mode="tensor", + required_sites=(), + ): + for value in (contract, execution, adapter): + require_sha(value) + if mode not in {"tensor", "metadata"}: + raise DiagnosticError("unsupported observation mode") + root.mkdir(mode=0o700, parents=True, exist_ok=False) + self.root, self.mode = root, mode + self.identities = {"contract": contract, "execution": execution, "adapter": adapter} + self.tensor_export, self.is_tensor = tensor_export, is_tensor + self.lock, self.next_id, self.closed = threading.RLock(), 0, False + self.next_observation = 0 + self.parent = contextvars.ContextVar(f"qwen-conformance-parent-{id(self)}", default=None) + self.rows, self.bindings, self.required = {}, {}, set(required_sites) + for site in self.required: + require_name(site) + + def bind(self, owner, name, *, site, hooks: HookSet, source_sha256=None, context=None): + require_name(site) + if site in self.bindings: + raise DiagnosticError("duplicate semantic call identity") + original = getattr(owner, name) + identity = callable_identity(original) + if source_sha256 is not None and identity != require_sha(source_sha256): + raise DiagnosticError("semantic call source changed") + self.bindings[site] = identity + registration = _custom_op_registration(original) if _is_custom_op(original) else None + + @functools.wraps(original) + def observed(*args, **kwargs): + if registration is not None and _custom_op_registration(original) != registration: + raise DiagnosticError("custom operator registration changed after source binding") + logical = context() if context is not None else {} + if logical is None: # explicitly unselected positions, no tensors read + return original(*args, **kwargs) + return self.invoke(site, original, args, kwargs, logical) + + hooks.replace(owner, name, observed) + + def _capture(self, root, values, logical): + with self.lock: + ordinal, self.next_observation = self.next_observation, self.next_observation + 1 + arrays, descriptors, unsupported = {}, {}, [] + + def visit(path, value): + if self.is_tensor(value): + # Metadata mode must not call tensor_export, .cpu(), .item(), + # synchronize(), or dereference a device pointer. + descriptors[path] = {"shape": list(value.shape), "dtype": str(value.dtype)} + if self.mode == "tensor": + arrays[path] = np.array(self.tensor_export(value), copy=True, order="C") + elif value is None or type(value) in (bool, int, float): + descriptors[path] = {"scalar": value} + elif isinstance(value, (tuple, list)): + descriptors[path] = {"container": type(value).__name__, "length": len(value)} + for i, child in enumerate(value): + visit(path + f".{i}", child) + elif isinstance(value, dict) and all(isinstance(k, str) for k in value): + descriptors[path] = {"keys": sorted(value)} + for key, child in sorted(value.items()): + require_name(key) + visit(path + "." + key, child) + else: + # Never repr() objects or strings: they can contain request text. + descriptors[path] = {"unexported_type": type(value).__qualname__} + unsupported.append(path) + + for path, value in values.items(): + visit(path, value) + frame = None + if arrays: + writer = FrameWriter( + root, + **self.identities, + input_digest=digest(logical), + phase="operator", + consumed=logical.get("consumed", 0), + pending=None, + expected=list(arrays), + logical={"capture": "call-tensors"}, + ) + for name, value in arrays.items(): + native_dtype = descriptors[name]["dtype"] + encoding = { + "torch.bfloat16": "bf16", + "torch.float8_e4m3fn": "fp8_e4m3fn", + "torch.float8_e4m3fnuz": "fp8_e4m3fnuz", + }.get(native_dtype) + if encoding is not None: + writer.add(name, value.tobytes(), dtype=encoding, shape=value.shape) + else: + writer.array(name, value) + frame = writer.finish()["sha256"] + return { + "frame": frame, + "descriptors": descriptors, + "unexported": unsupported, + "ordinal": ordinal, + } + + def invoke(self, site, function, args, kwargs, logical): + with self.lock: + if self.closed: + raise DiagnosticError("semantic recorder already finalized") + index, self.next_id = self.next_id, self.next_id + 1 + row = { + "index": index, + "site": site, + "parent": self.parent.get(), + "logical": logical, + "thread": threading.get_ident(), + "started_ns": time.monotonic_ns(), + } + self.rows[index] = row + path = self.root / f"call-{index:09d}" + path.mkdir(mode=0o700) + row["before"] = self._capture(path / "before", {"args": args, "kwargs": kwargs}, logical) + write_private(path / "started.json", seal(row)) + token = self.parent.set(index) + try: + result = function(*args, **kwargs) + row["after"] = self._capture( + path / "after", {"args": args, "kwargs": kwargs, "result": result}, logical + ) + row["completed"] = True + return result + except BaseException as exc: + row["completed"], row["exception_type"] = False, type(exc).__qualname__ + raise + finally: + self.parent.reset(token) + row["ended_ns"] = time.monotonic_ns() + write_private(path / "finished.json", seal(row)) + + def finish(self): + with self.lock: + if self.closed: + raise DiagnosticError("semantic recorder already finalized") + self.closed = True + if not self.rows or any(not r.get("completed") for r in self.rows.values()): + raise DiagnosticError("incomplete semantic call capture") + seen = {r["site"] for r in self.rows.values()} + if not self.required <= seen: + raise DiagnosticError("required semantic call was never observed") + rows = [seal(self.rows[i]) for i in range(self.next_id)] + doc = seal( + { + "schema": CALL_SCHEMA, + **self.identities, + "mode": self.mode, + "required_sites": sorted(self.required), + "bindings": self.bindings, + "calls": rows, + "device_binary_identity": "UNPROVED", + "non_tensor_state": "UNPROVED: explicit native state adapter required", + } + ) + write_private(self.root / "calls.json", doc) + return doc + + +def compare_calls(reference: Path, candidate: Path, output: Path): + docs = [private_json(p / "calls.json") for p in (reference, candidate)] + for doc in docs: + authenticate(doc) + for name in ("contract", "execution", "adapter"): + require_sha(doc[name]) + if doc.get("schema") != CALL_SCHEMA or doc.get("mode") != "tensor" or not doc.get("calls"): + raise DiagnosticError("exact call comparison requires complete tensor evidence") + if not set(doc["required_sites"]) <= {r["site"] for r in doc["calls"]}: + raise DiagnosticError("call trace omitted required sites") + if not isinstance(doc.get("bindings"), dict) or not doc["bindings"]: + raise DiagnosticError("semantic call source bindings are missing") + for name, identity in doc["bindings"].items(): + require_name(name) + require_sha(identity) + if docs[0]["contract"] != docs[1]["contract"] or len(docs[0]["calls"]) != len(docs[1]["calls"]): + raise DiagnosticError("different semantic contracts or call schedules") + first, tensors, observations = None, 0, [] + for i, (a, b) in enumerate(zip(docs[0]["calls"], docs[1]["calls"], strict=True)): + for root, row, doc in ((reference, a, docs[0]), (candidate, b, docs[1])): + authenticate(row) + if row["index"] != i or not row["completed"]: + raise DiagnosticError("reordered or incomplete semantic call") + recorded = private_json(root / f"call-{i:09d}" / "finished.json") + if row != recorded: + raise DiagnosticError("semantic call receipt changed") + if row["site"] not in doc["bindings"]: + raise DiagnosticError("observed call is not bound to an implementation") + started = private_json(root / f"call-{i:09d}" / "started.json") + authenticate(started) + if any(row.get(k) != v for k, v in started.items() if k != "sha256"): + raise DiagnosticError("call changed its captured input receipt") + if any(a[k] != b[k] for k in ("site", "parent", "logical")): + raise DiagnosticError("call boundaries need an explicit logical adapter") + for phase in ("before", "after"): + ca, cb = a[phase], b[phase] + if ca["ordinal"] != cb["ordinal"]: + raise DiagnosticError("different causal observation order") + observations.append((ca["ordinal"], i, a["site"], phase, ca, cb)) + if sorted(r[0] for r in observations) != list(range(len(observations))): + raise DiagnosticError("missing or duplicate call observations") + for _, i, site, phase, ca, cb in sorted(observations): + if ca["descriptors"] != cb["descriptors"] and first is None: + first = {"call": i, "site": site, "phase": phase, "kind": "metadata"} + if ca["unexported"] or cb["unexported"]: + raise DiagnosticError("call contains unexported state; add its explicit adapter") + if bool(ca["frame"]) != bool(cb["frame"]): + raise DiagnosticError("call tensor coverage differs") + if ca["frame"]: + roots = [p / f"call-{i:09d}" / phase for p in (reference, candidate)] + for root, capture, doc in zip(roots, (ca, cb), docs, strict=True): + frame = read_frame(root) + if frame["sha256"] != capture["frame"]: + raise DiagnosticError("call tensors changed after capture") + if any(frame[k] != doc[k] for k in ("contract", "execution", "adapter")): + raise DiagnosticError("call tensors belong to a different execution") + result = compare_frames(*roots) + tensors += 1 + if not result["equal"] and first is None: + first = { + "call": i, + "site": site, + "phase": phase, + "difference": result["first_difference"], + } + if tensors == 0: + raise DiagnosticError("no numerical observations in call comparison") + report = seal( + { + "schema": "urn:qwen:call-comparison:v1", + "equal": first is None, + "first_difference": first, + "calls": len(docs[0]["calls"]), + "status": "TESTED", + "scope": "captured call tensors and metadata", + "native_equivalence": "UNPROVED", + } + ) + write_private(output, report) + return report diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_invariants.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_invariants.py new file mode 100644 index 0000000..fa2282d --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_invariants.py @@ -0,0 +1,123 @@ +"""Executable control/decision formulae used by the conformance checker. + +These small expressions are also called with symbolic arguments by Z3. Their +proofs cover these functions, not a GPU implementation that merely resembles +them. Numeric certificates are conditional on independently justified bounds. +""" + +from fractions import Fraction + +from qwen_r9700_lab.conformance_gate import publication_allowed +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, integer, seal + + +def transaction_ready(output, state, identity, coverage, completed, cancelled, base, current): + return ( + publication_allowed(output, state, identity, coverage) + & completed + & (cancelled == False) # noqa: E712 -- works on Python and symbolic Booleans + & (base == current) + ) + + +def version_matches(position, kv, gdn, conv): + return (position == kv) & (position == gdn) & (position == conv) + + +def writable_exclusively(references, external_references, immutable): + return ( + (references == 1) & (external_references == 0) & (immutable == False) # noqa: E712 + ) + + +def restore_matches( + contract, restored_contract, generation, restored_generation, size, restored_size +): + return ( + (contract == restored_contract) + & (generation == restored_generation) + & (size == restored_size) + ) + + +def snapshot_publishable(verified, durable, current, candidate): + return verified & durable & (current == candidate) + + +def separated_intervals(winner_lower, competitor_upper): + return winner_lower > competitor_upper + + +def unresolved_frontiers_separated( + cutoff, local_upper, global_upper, local_unresolved, global_unresolved +): + """Compare already conservative bounds for both unrescored partitions. + + Local means dropped within a vocabulary block; global means emitted by + that block but omitted from exact reranking. Empty partitions are ignored. + This predicate does not establish coverage, score identity or bounds. + No rounding-sensitive arithmetic is performed here. Native bound + construction, cutoff selection and invocation binding remain unproved. + """ + return ( + (local_unresolved == False) | (cutoff > local_upper) # noqa: E712 + ) & ( + (global_unresolved == False) | (cutoff > global_upper) # noqa: E712 + ) + + +def margin_preserved(margin, error): + return (error >= 0) & (margin > 2 * error) + + +def interval_certificate(lower, upper, winner: int, *, bound_origin: str) -> dict: + """Exact rational inequalities; never turn empirical errors into a proof. + + All vocabulary entries must be present. Missing/excluded logits cannot be + assumed harmless merely because an INT2 shortlist was exactly reranked. + The caller supplies sound intervals including upstream state/hidden error. + """ + integer(winner) + if not lower or len(lower) != len(upper) or winner >= len(lower) or not bound_origin: + raise DiagnosticError("incomplete full-vocabulary interval certificate") + lo, hi = [Fraction(v) for v in lower], [Fraction(v) for v in upper] + if any(a > b for a, b in zip(lo, hi, strict=True)): + raise DiagnosticError("reversed logit interval") + certified = all(separated_intervals(lo[winner], hi[i]) for i in range(len(lo)) if i != winner) + return seal( + { + "schema": "urn:qwen:conditional-argmax-certificate:v1", + "vocabulary": len(lo), + "winner": winner, + "certified_under_bounds": certified, + "lower": [str(v) for v in lo], + "upper": [str(v) for v in hi], + "bound_origin": bound_origin, + "bounds_soundness": "ASSUMED", + "status": "RUNTIME-CHECKED", + "backend_equivalence": "UNPROVED", + } + ) + + +def rejection_distribution(target, draft): + """Exact finite-distribution reference for one speculative rejection step. + + P(token) = min(p,q) + Z * max(p-q,0)/Z = p. Z=0 means + all proposals are accepted; no residual distribution is sampled. + Does not specify RNG consumption or imply equal seeded sample paths. + """ + p, q = [Fraction(v) for v in target], [Fraction(v) for v in draft] + if not p or len(p) != len(q) or sum(p) != 1 or sum(q) != 1 or any(v < 0 for v in p + q): + raise DiagnosticError("invalid complete target/draft probability distribution") + accepted = [min(a, b) for a, b in zip(p, q, strict=True)] + residual = [a - b for a, b in zip(p, accepted, strict=True)] + rejected_mass = sum(residual) + conditional = [v / rejected_mass for v in residual] if rejected_mass else None + final = [a + r for a, r in zip(accepted, residual, strict=True)] + return { + "accepted_mass": accepted, + "rejected_mass": rejected_mass, + "residual": conditional, + "output": final, + } diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_lifecycle.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_lifecycle.py new file mode 100644 index 0000000..37d207a --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_lifecycle.py @@ -0,0 +1,335 @@ +"""Recovery comparisons and real compressed-store transport, with no GPU imports. + +Transport tests exercise Radiance's ChatStore, not a substitute JSON snapshot. +The canonical-frame envelope is diagnostic: it does not claim to implement the +native vLLM connector's page mapping. Native captures use the same comparison +contract through RecoveryCapture; their extraction must be qualified separately. +""" + +from __future__ import annotations + +import hashlib +import json +import struct +import threading +from pathlib import Path + +from qwen_r9700_lab.conformance_state import ( + FrameWriter, + archive_frame, + compare_frames, + open_blob, + read_frame, +) +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + integer, + require_sha, + seal, + write_private, +) +from qwen_r9700_lab.radiance_cache import KEY, ChatStore + +ENVELOPE = b"QWENFRAME1\0" +RECOVERY_SCHEMA = "urn:qwen:recovery-comparison:v1" +STAGES = ("reference", "live", "restored") + + +def state_view(source: Path, destination: Path, required: list[str]) -> dict: + """Compare logical state across phases, excluding outputs and physical layout. + + Required coverage comes from the model contract, never from an intersection + of whatever two producers happen to emit. All original evidence is retained. + """ + frame = read_frame(source) + if not required or len(set(required)) != len(required): + raise DiagnosticError("recovery comparison requires explicit complete state coverage") + if any(name not in frame["components"] for name in required): + raise DiagnosticError("recovery capture omitted required state") + writer = FrameWriter( + destination, + contract=frame["contract"], + execution=frame["execution"], + adapter=frame["adapter"], + input_digest=frame["input_digest"], + phase="restore", + consumed=frame["consumed"], + pending=frame["pending"], + expected=required, + # Mode describes the producer. All other logical metadata is compared; + # dropping sampler/version fields would hide latent state corruption. + logical={k: v for k, v in frame["logical"].items() if k != "execution_mode"}, + ) + for name in required: + descriptor = frame["components"][name] + with open_blob(source, descriptor["file"]) as stream: + writer.add_stream( + name, + iter(lambda: stream.read(1024 * 1024), b""), + dtype=descriptor["dtype"], + shape=descriptor["shape"], + ) + if writer.components[name]["sha256"] != descriptor["sha256"]: + raise DiagnosticError("recovery source changed during capture") + return writer.finish() + + +def recovery_classification(live_equal: bool, restored_equal: bool, unchanged: bool) -> str: + if live_equal and restored_equal: + return "all_observed_state_matches" + if not live_equal and restored_equal: + return "live_differs_restored_matches_reference" + if live_equal and not restored_equal: + return "restore_introduced_difference" + return "same_difference_survives_restore" if unchanged else "both_differ_from_reference" + + +def compare_recovery( + reference: Path, live: Path, restored: Path, output: Path, *, required: list[str] +) -> dict: + output.mkdir(mode=0o700, parents=True, exist_ok=False) + sources = dict(zip(STAGES, (reference, live, restored), strict=True)) + identities = {} + for stage, path in sources.items(): + archived = output / (stage + "-original") + identities[stage] = archive_frame(path, archived)["sha256"] + state_view(archived, output / stage, required) + comparisons = { + "reference_live": compare_frames(output / "reference", output / "live"), + "reference_restored": compare_frames(output / "reference", output / "restored"), + "live_restored": compare_frames(output / "live", output / "restored"), + } + report = seal( + { + "schema": RECOVERY_SCHEMA, + "sources": identities, + "required_components": required, + "comparisons": comparisons, + "classification": recovery_classification( + comparisons["reference_live"]["equal"], + comparisons["reference_restored"]["equal"], + comparisons["live_restored"]["equal"], + ), + "equal": all(r["equal"] for r in comparisons.values()), + "status": "TESTED", + "scope": "captured logical state at identical input prefix and position", + "corruption_cause": "UNPROVED: a numerical difference is not itself a causal diagnosis", + "reference_correctness": "ASSUMED: independently qualify the reference producer", + "native_connector_qualification": "UNPROVED", + } + ) + write_private(output / "report.json", report) + return report + + +def pack_frame(source: Path) -> bytes: + frame = read_frame(source) + if not compare_frames(source, source)["equal"]: + raise DiagnosticError("nonfinite snapshot cannot enter a checked transport") + metadata = json.dumps(frame, sort_keys=True, separators=(",", ":")).encode() + parts = [ENVELOPE, struct.pack(" dict: + header = len(ENVELOPE) + 8 + if len(payload) < header or payload[: len(ENVELOPE)] != ENVELOPE: + raise DiagnosticError("invalid diagnostic snapshot envelope") + length = struct.unpack_from(" len(payload) - header: + raise DiagnosticError("truncated diagnostic snapshot header") + try: + frame = json.loads(payload[header : header + length]) + authenticate(frame) + coverage = frame["coverage"] + components = frame["components"] + if not isinstance(coverage, list) or set(coverage) != set(components): + raise DiagnosticError("invalid diagnostic snapshot coverage") + total = sum(integer(components[n]["nbytes"]) for n in coverage) + if header + length + total != len(payload): + raise DiagnosticError("snapshot payload has missing or extra bytes") + except (KeyError, TypeError, ValueError) as exc: + raise DiagnosticError("invalid diagnostic snapshot metadata") from exc + writer = FrameWriter( + output, + **{k: frame[k] for k in ("contract", "execution", "adapter", "input_digest")}, + phase=frame["phase"], + consumed=frame["consumed"], + pending=frame["pending"], + expected=coverage, + logical=frame["logical"], + ) + cursor = header + length + for name in coverage: + descriptor = components[name] + data = payload[cursor : cursor + descriptor["nbytes"]] + cursor += descriptor["nbytes"] + if hashlib.sha256(data).hexdigest() != descriptor["sha256"]: + raise DiagnosticError("diagnostic snapshot payload checksum mismatch") + writer.add(name, data, dtype=descriptor["dtype"], shape=descriptor["shape"]) + result = writer.finish() + if result != frame: + raise DiagnosticError("snapshot roundtrip changed its canonical identity") + return result + + +class FrameTransport: + """Immutable RAM copies plus real ChatStore compression/publication/GC. + + A content-addressed receipt is returned only after verified publication. + Callers keep the previous receipt until then. Restart constructs a new + transport with an empty RAM bank and restores through ChatStore.read. + """ + + def __init__(self, root: Path, chat: dict, *, block_size=4096): + if integer(block_size) < 64: + raise DiagnosticError("diagnostic snapshot blocks are too small") + self.store = ChatStore(root, chat) + self.block_size = block_size + self.ram: bytes | None = None + self.store.activate() + + def save_ram(self, source: Path) -> None: + self.ram = pack_frame(source) + + def restore_ram(self, output: Path) -> dict: + if self.ram is None: + raise DiagnosticError("RAM snapshot was evicted") + return unpack_frame(self.ram, output) + + def stage_disk(self, source: Path) -> dict: + payload = pack_frame(source) + keys, blocks = [], [] + for offset in range(0, len(payload), self.block_size): + data = payload[offset : offset + self.block_size].ljust(self.block_size, b"\0") + key = "g0-" + hashlib.sha256(data).hexdigest() + ".qkv" + keys.append(key) + blocks.append((key, memoryview(data))) + if not self.store.write_many(blocks): + raise DiagnosticError("snapshot generation was superseded while staging") + frame = read_frame(source) + return seal( + { + "schema": "urn:qwen:diagnostic-frame-receipt:v1", + "chat": self.store.chat, + "keys": keys, + "block_size": self.block_size, + "payload_bytes": len(payload), + "payload_sha256": hashlib.sha256(payload).hexdigest(), + "frame_sha256": frame["sha256"], + "consumed": frame["consumed"], + } + ) + + def publish(self, receipt: dict) -> bool: + self._validate_receipt(receipt) + return self.store.publish(receipt["keys"], receipt["consumed"], self.block_size) + + def save_disk(self, source: Path) -> dict: + receipt = self.stage_disk(source) + if not self.publish(receipt): + raise DiagnosticError("incomplete checkpoint: previous disk head retained") + return receipt + + def _validate_receipt(self, receipt): + authenticate(receipt) + if ( + receipt.get("schema") != "urn:qwen:diagnostic-frame-receipt:v1" + or receipt.get("chat") != self.store.chat + or receipt.get("block_size") != self.block_size + or not receipt.get("keys") + ): + raise DiagnosticError("snapshot receipt belongs to a different chat or generation") + for key in receipt["keys"]: + if not isinstance(key, str) or not KEY.fullmatch(key): + raise DiagnosticError("invalid diagnostic snapshot object key") + require_sha(receipt["payload_sha256"]) + require_sha(receipt["frame_sha256"]) + size = integer(receipt["payload_bytes"]) + if ( + not (len(receipt["keys"]) - 1) * self.block_size + < size + <= len(receipt["keys"]) * self.block_size + ): + raise DiagnosticError("snapshot receipt has inconsistent length") + + def restore_disk(self, receipt: dict, output: Path) -> dict: + import zstandard + + self._validate_receipt(receipt) + # Hold one shared lock over the entire read: a new head cannot collect + # some of these blocks halfway through the restored frame. + try: + payload = b"".join(self.store.read_many(receipt["keys"], self.block_size)) + except zstandard.ZstdError as exc: + raise DiagnosticError("compressed snapshot failed integrity verification") from exc + data = payload[: receipt["payload_bytes"]] + if any(payload[receipt["payload_bytes"] :]): + raise DiagnosticError("snapshot padding changed") + if hashlib.sha256(data).hexdigest() != receipt["payload_sha256"]: + raise DiagnosticError("snapshot receipt does not match restored bytes") + result = unpack_frame(data, output) + if result["sha256"] != receipt["frame_sha256"]: + raise DiagnosticError("snapshot receipt does not match restored frame") + return result + + def evict_ram(self, source: Path) -> dict: + receipt = self.save_disk(source) + self.ram = None # publication must succeed before the last RAM copy goes away + return receipt + + +class RecoveryCapture: + """Adapter-neutral, immutable capture points for real native lifecycle tests. + + A native producer supplies a complete frame at an already committed boundary. + This class never invokes a GPU operation or tries to guess a page layout. + """ + + def __init__(self, root: Path, *, contract: str, chat: str, generation: str, required): + for value in (contract, chat, generation): + require_sha(value) + root.mkdir(mode=0o700, parents=True, exist_ok=False) + self.root, self.required = root, list(required) + self.identity = {"contract": contract, "chat": chat, "generation": generation} + self.lock = threading.Lock() + self.frames = {} + + def record(self, stage: str, source: Path, *, chat: str, generation: str, quiescent: bool): + with self.lock: + if stage not in STAGES or stage in self.frames: + raise DiagnosticError("duplicate or unregistered recovery capture stage") + if chat != self.identity["chat"] or generation != self.identity["generation"]: + raise DiagnosticError("recovery crossed a chat or compaction generation") + if quiescent is not True: + raise DiagnosticError("full state capture needs a completed state transition") + frame = read_frame(source) + if frame["contract"] != self.identity["contract"]: + raise DiagnosticError("recovery capture changed numerical contract") + if not set(self.required).issubset(frame["coverage"]): + raise DiagnosticError("recovery capture is missing required state") + self.frames[stage] = archive_frame(source, self.root / stage)["sha256"] + write_private( + self.root / (stage + "-receipt.json"), + seal({**self.identity, "stage": stage, "frame": self.frames[stage]}), + ) + + def compare(self): + if set(self.frames) != set(STAGES): + raise DiagnosticError( + "recovery comparison missing a capture; never infer a clean state" + ) + return compare_recovery( + *(self.root / s for s in STAGES), self.root / "comparison", required=self.required + ) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_mode_boundaries.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_mode_boundaries.py new file mode 100644 index 0000000..2526674 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_mode_boundaries.py @@ -0,0 +1,256 @@ +"""Align observed semantic boundaries without claiming complete state coverage. + +Event numbers and physical addresses are not semantic identities. Compare only +unique named operations with the same logical owner and ordered positions. +Tensor equality here is not vocabulary top-k or proof of a complete transition. +""" + +from collections import defaultdict + +import numpy as np + +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, authenticate, seal + + +def silu_cut(metadata, tensors, layer): + """Identify the MLP activation by both adjacent named projections and wiring.""" + authenticate(metadata) + stem = f"language_model.model.layers.{layer}.mlp." + + def projection(suffix): + matches = [ + event + for event in metadata["events"] + if event["operation"] == "radiance.mxfp4_linear.default" + and any( + owner.startswith(stem + suffix + ".") + for owner in event.get("logical_identities", ()) + ) + ] + if len(matches) != 1: + raise DiagnosticError("MLP cut needs unique same-layer projection owners") + return matches[0] + + gate, down = projection("gate_up_proj"), projection("down_proj") + middle = [e for e in metadata["events"] if gate["index"] < e["index"] < down["index"]] + if len(middle) != 1: + raise DiagnosticError("unrecognized operation sequence inside the MLP activation cut") + activation = middle[0] + name = activation["operation"] + if name == "_C.silu_and_mul.default": + input_key = f"{activation['index']}.before.args.1" + output_key = f"{activation['index']}.after.mutable.result" + elif name in { + "inductor/triton_poi_fused_mul_mxfp4_linear_silu_slice_0", + "inductor/triton_poi_fused_mul_mxfp4_linear_silu_slice_1", + }: + input_key = f"{activation['index']}.before.args.0" + output_key = f"{activation['index']}.after.out_ptr0" + else: + raise DiagnosticError("unrecognized MLP activation implementation") + descriptors = { + r["key"]: r + for e in (gate, activation, down) + for phase in ("before", "after") + for r in e[phase] + } + gate_key, down_key = f"{gate['index']}.after.result", f"{down['index']}.before.args.0" + keys = (input_key, output_key, gate_key, down_key) + if any(k not in descriptors or k not in tensors for k in keys): + raise DiagnosticError("MLP cut lacks captured inputs or outputs") + if any(descriptors[k]["dtype"] != "torch.bfloat16" for k in keys): + raise DiagnosticError("MLP cut requires the pinned BF16 activation contract") + positions = metadata["positions"] + if descriptors[input_key]["shape"] != [len(positions), 34816] or descriptors[output_key][ + "shape" + ] != [len(positions), 17408]: + raise DiagnosticError("MLP activation shape differs from this pinned model") + for left, right in ((input_key, gate_key), (output_key, down_key)): + result = compare_arrays([tensors[left]], [tensors[right]], positions) + if result is None or result["exact_positions"] != len(positions): + raise DiagnosticError( + "MLP activation tensors do not connect to their named projections" + ) + return {"operation": name, "input": input_key, "output": output_key} + + +def anchors(metadata): + authenticate(metadata) + grouped = defaultdict(list) + for event in metadata["events"]: + identities = tuple(event.get("logical_identities", ())) + if identities and event["after"]: + grouped[(event["operation"], identities)].append(event) + # Repeated calls with the same owner may have different semantic roles. + # Leave these unpaired instead of assuming the first occurrence corresponds. + unique = {key: value[0] for key, value in grouped.items() if len(value) == 1} + return unique, len(grouped) - len(unique) + + +def compare_arrays(left, right, positions): + """Compare ordered captured tensors; return unknown for incomplete layouts.""" + if not left or len(left) != len(right) or not positions: + return None + equal = np.ones(len(positions), dtype=np.bool_) + elements = 0 + differing = 0 + for a, b in zip(left, right, strict=True): + if a.shape != b.shape or a.dtype != b.dtype or not a.ndim: + return None + if a.shape[0] not in (len(positions), len(positions) * 48): + return None + # Byte comparison retains signed zero/NaN payload and FP8 storage bits. + raw_a = np.ascontiguousarray(a).view(np.uint8).reshape(len(positions), -1) + raw_b = np.ascontiguousarray(b).view(np.uint8).reshape(len(positions), -1) + equal &= (raw_a == raw_b).all(axis=1) + elements += a.size + differing += np.count_nonzero((raw_a != raw_b).reshape(-1, a.dtype.itemsize).any(axis=1)) + return { + "positions": len(positions), + "exact_positions": int(equal.sum()), + "different_positions": [p for p, ok in zip(positions, equal, strict=True) if not ok], + "compared_elements": int(elements), + "different_elements": int(differing), + } + + +def admit_bridge(bridge, pass_receipt, capture, prefill=None): + """Bind tensors to a completed pass whose observer reproduced its control.""" + for value in (bridge, pass_receipt, capture): + authenticate(value) + admission = bridge["admission"] + authenticate(admission) + if sorted(admission["captures"]) != [False, True]: + raise DiagnosticError("output bridge must compare an observer with an uncaptured control") + side = admission["captures"].index(True) + if admission["receipts"][side][-1] != pass_receipt["sha256"]: + raise DiagnosticError("output bridge belongs to another captured pass") + if ( + bridge["decode"]["positions"] != 320 + or bridge["decode"]["full_logits_exact"] != 320 + or bridge["prefill"]["positions"] != 1 + or bridge["prefill"]["full_logits_exact"] != 1 + ): + raise DiagnosticError("observer changed the prefill or decode outputs") + observed = pass_receipt["observation"]["isolated_capture"] + if observed["sha256"] != capture["sha256"]: + raise DiagnosticError("tensor capture does not belong to the bridged pass") + if prefill is not None: + authenticate(prefill) + if ( + prefill["sha256"] != observed["prefill_capture"] + or prefill["decode_capture"] != capture["sha256"] + ): + raise DiagnosticError("sampled prefill capture does not belong to the bridged pass") + + +def compare_group(left, right, left_tensors, right_tensors): + a, aa = anchors(left) + b, ab = anchors(right) + positions = left["positions"] + if not positions or positions != right["positions"] or len(set(positions)) != len(positions): + raise DiagnosticError("boundary captures have different or duplicate positions") + records = [] + for key in sorted(a.keys() & b.keys(), key=lambda k: a[k]["index"]): + ea, eb = a[key], b[key] + + def values(event, tensors, which): + return [tensors[r["key"]] for r in event[which]] + + def layouts(event, which): + return [(r.get("shape"), r.get("dtype")) for r in event[which]] + + if layouts(ea, "after") != layouts(eb, "after"): + continue + after = compare_arrays( + values(ea, left_tensors, "after"), values(eb, right_tensors, "after"), positions + ) + if after is None: + continue + before = compare_arrays( + values(ea, left_tensors, "before"), values(eb, right_tensors, "before"), positions + ) + if layouts(ea, "before") != layouts(eb, "before"): + before = None + records.append( + { + "operation": key[0], + "owners": list(key[1]), + "left_event": ea["index"], + "right_event": eb["index"], + "captured_inputs": before, + "outputs": after, + } + ) + if not records: + raise DiagnosticError("no unambiguous comparable semantic boundaries") + return { + "positions": positions, + "boundaries": records, + "left_unique_unpaired": len(a.keys() - b.keys()), + "right_unique_unpaired": len(b.keys() - a.keys()), + "left_ambiguous": aa, + "right_ambiguous": ab, + } + + +def summarize(groups, *, expected_positions, sources): + observed = [p for group in groups for p in group["positions"]] + if observed != expected_positions or len(set(observed)) != len(observed): + raise DiagnosticError("boundary comparison has incomplete or duplicate position coverage") + totals = {} + first = None + for group in groups: + for row in group["boundaries"]: + key = (row["operation"], tuple(row["owners"])) + total = totals.setdefault( + key, + { + "operation": key[0], + "owners": list(key[1]), + "positions": 0, + "exact_output_positions": 0, + "different_output_elements": 0, + "captured_input_positions": 0, + "exact_captured_input_positions": 0, + }, + ) + output, inputs = row["outputs"], row["captured_inputs"] + total["positions"] += output["positions"] + total["exact_output_positions"] += output["exact_positions"] + total["different_output_elements"] += output["different_elements"] + if inputs is not None: + total["captured_input_positions"] += inputs["positions"] + total["exact_captured_input_positions"] += inputs["exact_positions"] + if output["different_positions"]: + candidate = { + "position": min(output["different_positions"]), + "operation": key[0], + "owners": list(key[1]), + "left_event": row["left_event"], + "right_event": row["right_event"], + "all_captured_inputs_exact": inputs is not None + and inputs["exact_positions"] == inputs["positions"], + } + if first is None or (candidate["position"], candidate["left_event"]) < ( + first["position"], + first["left_event"], + ): + first = candidate + return seal( + { + "schema": "qwen.execution-mode-boundary-comparison.v1", + "status": "COMPARED_OBSERVED_BOUNDARIES", + "sources": sources, + "positions": len(observed), + "boundaries": list(totals.values()), + "first_observed_different_boundary": first, + "unpaired_and_ambiguous": [ + {k: v for k, v in g.items() if k not in ("positions", "boundaries")} for g in groups + ], + "scope": ( + "Named captured activation boundaries only; intervening operations and full " + "KV/GDN/conv state are not proved equal. No isolated vocabulary top-k claim." + ), + } + ) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_model.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_model.py new file mode 100644 index 0000000..15fe3a1 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_model.py @@ -0,0 +1,476 @@ +"""Standalone CPU reference for text-only dense Qwen3.5 MXFP4 checkpoints. + +Uses checkpoint order directly, without native weight permutation, speculative +execution, fused kernels or a vLLM-generated initial state. Every token passes +through the serial finite-precision operators in conformance_reference. +""" + +from __future__ import annotations + +import hashlib +import json +import math +import mmap +import os +import stat +import struct +from collections.abc import Callable, Mapping +from pathlib import Path + +import numpy as np + +from qwen_r9700_lab import conformance_reference as ref +from qwen_r9700_lab.conformance_state import FrameWriter, load_arrays +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, digest, integer + +STORAGE_DTYPES = { + "U8": "|u1", + "I8": "|i1", + "I32": " 64 * 1024 * 1024 or length + 8 > len(mapped): + mapped.close() + handle.close() + raise DiagnosticError("invalid safetensors header") + self.handles[shard], self.maps[shard] = handle, mapped + self.headers[shard] = (8 + length, json.loads(mapped[8 : 8 + length])) + offset, headers = self.headers[shard] + descriptor = headers[name] + dtype = STORAGE_DTYPES.get(descriptor["dtype"]) + if dtype is None: + raise DiagnosticError("checkpoint dtype is outside this reference contract") + storage_dtype = " len(self.maps[shard]) + ): + raise DiagnosticError("checkpoint tensor is outside its shard") + if end - begin != math.prod(shape) * np.dtype(storage_dtype).itemsize: + raise DiagnosticError("checkpoint tensor size mismatch") + array = np.ndarray( + shape, dtype=storage_dtype, buffer=self.maps[shard], offset=offset + begin + ) + if rows is not None: + array = array[rows] + if dtype == "bf16": + return (array.astype(np.uint32) << 16).view(np.float32) + return np.array(array, copy=True) + + def close(self): + for value in self.maps.values(): + value.close() + for value in self.handles.values(): + value.close() + self.maps.clear() + self.handles.clear() + + +def state_names(config: Mapping) -> list[str]: + names = ["sequence.tokens", "sequence.position"] + for layer, kind in enumerate(config["layer_types"]): + prefix = f"layer.{layer:03d}." + names.extend( + prefix + suffix + for suffix in ( + ("gdn", "conv") if kind == "linear_attention" else ("keys", "values", "kv_scales") + ) + ) + return names + + +class QuantizedQwenReference: + def __init__( + self, + checkpoint, + *, + kv_scales: Mapping[str, list[float]], + contract: str, + execution: str, + adapter: str, + capture: Callable | None = None, + reference_profile: str = "radiance-fp8", + ): + self.weights = checkpoint + self.precision = ref.reference_precision(reference_profile) + self.config = checkpoint.config.get("text_config", checkpoint.config) + c = self.config + if c["model_type"] != "qwen3_5_text" or c.get("layer_scale", False): + raise DiagnosticError("reference supports dense text-only Qwen3.5 without layer scales") + if c.get("hidden_act", "silu") != "silu" or c.get("attention_bias", False): + raise DiagnosticError("unsupported activation or attention bias") + if c.get("rope_parameters", {}).get("rope_type", "default") != "default": + raise DiagnosticError("reference requires the default text rotary contract") + if len(c["layer_types"]) != c["num_hidden_layers"] or any( + k not in {"linear_attention", "full_attention"} for k in c["layer_types"] + ): + raise DiagnosticError("unsupported layer inventory") + kv_scales = ref.kv_scaling(c, dict(kv_scales), reference_profile) + self.kv_scales = {k: np.asarray(v, dtype=np.float32) for k, v in kv_scales.items()} + self.contract, self.execution, self.adapter = contract, execution, adapter + self.capture = capture + self._linear = ref.linear + self.tokens, self.state = [], {} + self.prefix = "model.language_model." + self.reset() + + def observe(self, position, layer, name, value): + if self.capture is not None: + self.capture(position, layer, name, np.array(value, copy=True)) + + def reset(self): + c = self.config + self.tokens, self.state = [], {} + for layer, kind in enumerate(c["layer_types"]): + if kind == "linear_attention": + heads, kd, vd = ( + c["linear_num_value_heads"], + c["linear_key_head_dim"], + c["linear_value_head_dim"], + ) + channels = 2 * c["linear_num_key_heads"] * kd + heads * vd + self.state[layer] = { + "gdn": np.zeros((heads, vd, kd), dtype=np.float32), + "conv": np.zeros((channels, c["linear_conv_kernel_dim"] - 1), dtype=np.float32), + } + else: + shape = (0, c["num_key_value_heads"], c["head_dim"]) + self.state[layer] = { + "keys": np.empty(shape, dtype="u1" if self.precision["kv_fp8"] else "= c["vocab_size"] or position >= c["max_position_embeddings"]: + raise DiagnosticError("token or position is outside the model contract") + x = self.weights.tensor(self.prefix + "embed_tokens.weight", slice(token, token + 1))[0] + x = ref.bf16(x) + for layer, kind in enumerate(c["layer_types"]): + base = self.prefix + f"layers.{layer}." + self.observe(position, layer, "input", x) + normed = ref.rms_norm( + x, + self.weights.tensor(base + "input_layernorm.weight"), + c["rms_norm_eps"], + weight_offset=1, + ) + self.observe(position, layer, "input_norm", normed) + if kind == "linear_attention": + out = self.gdn(layer, base + "linear_attn.", normed, position) + else: + out = self.attention(layer, base + "self_attn.", normed, position) + x = ref.bf16(np.add(x, out, dtype=np.float32)) + self.observe(position, layer, "attention_residual", x) + normed = ref.rms_norm( + x, + self.weights.tensor(base + "post_attention_layernorm.weight"), + c["rms_norm_eps"], + weight_offset=1, + ) + self.observe(position, layer, "post_attention_norm", normed) + gate = self.project(base + "mlp.gate_proj", normed) + up = self.project(base + "mlp.up_proj", normed) + self.observe(position, layer, "mlp_gate", gate) + self.observe(position, layer, "mlp_up", up) + intermediate = ref.bf16(ref.silu(gate) * up) + self.observe(position, layer, "mlp_intermediate", intermediate) + down = self.project(base + "mlp.down_proj", intermediate) + self.observe(position, layer, "mlp_down", down) + x = ref.bf16(x + down) + self.observe(position, layer, "output", x) + self.tokens.append(token) + hidden = ref.rms_norm( + x, self.weights.tensor(self.prefix + "norm.weight"), c["rms_norm_eps"], weight_offset=1 + ) + logits = self.project("lm_head", hidden).astype(np.float32) + self.observe(position, -1, "logits", logits) + if not np.isfinite(logits).all(): + raise DiagnosticError("nonfinite reference logits") + return logits + + def gdn(self, layer, base, x, position): + c, state = self.config, self.state[layer] + kh, vh = c["linear_num_key_heads"], c["linear_num_value_heads"] + kd, vd = c["linear_key_head_dim"], c["linear_value_head_dim"] + qkv = self.project(base + "in_proj_qkv", x) + z = self.project(base + "in_proj_z", x).reshape(vh, vd) + b, a = self.project(base + "in_proj_b", x), self.project(base + "in_proj_a", x) + for name, value in ( + ("gdn_qkv", qkv), + ("gdn_z", z), + ("gdn_b", b), + ("gdn_a", a), + ("conv_input_state", state["conv"]), + ): + self.observe(position, layer, name, value) + conv_weight = self.weights.tensor(base + "conv1d.weight").reshape(qkv.size, -1) + conv, state["conv"] = ref.convolution_step(qkv, state["conv"], conv_weight) + self.observe(position, layer, "conv_output", conv) + self.observe(position, layer, "conv_state", state["conv"]) + q, k, v = np.split(conv, (kh * kd, 2 * kh * kd)) + q, k, v = q.reshape(kh, kd), k.reshape(kh, kd), v.reshape(vh, vd) + q, k = ( + ref.bf16(t / np.sqrt(ref.ordered_sum(t * t)[..., None] + np.float32(1e-6))) + for t in (q, k) + ) + decay = -np.exp(self.weights.tensor(base + "A_log").astype(np.float32)) * ref.softplus( + a + self.weights.tensor(base + "dt_bias") + ) + beta = ref.bf16(ref.sigmoid(b)) + for name, value in ( + ("gdn_q", q), + ("gdn_k", k), + ("gdn_v", v), + ("gdn_decay", decay), + ("gdn_beta", beta), + ("gdn_input_state", state["gdn"]), + ): + self.observe(position, layer, name, value) + out, state["gdn"] = ref.gdn_step(q, k, v, decay, beta, state["gdn"]) + self.observe(position, layer, "gdn_output", out) + self.observe(position, layer, "gdn_state", state["gdn"]) + gated = ref.rms_norm( + out, self.weights.tensor(base + "norm.weight"), c["rms_norm_eps"], output_bf16=False + ) * ref.silu(z) + self.observe(position, layer, "gdn_gated", gated) + projected = self.project(base + "out_proj", ref.bf16(gated).reshape(-1)) + self.observe(position, layer, "attention_projected", projected) + return projected + + def attention(self, layer, base, x, position): + c, state = self.config, self.state[layer] + heads, kv_heads, width = c["num_attention_heads"], c["num_key_value_heads"], c["head_dim"] + q_raw = self.project(base + "q_proj", x) + gate = None + if c.get("attn_output_gate", True): + q, gate = np.split(q_raw.reshape(heads, width * 2), 2, axis=-1) + else: + q = q_raw.reshape(heads, width) + k, v = ( + self.project(base + name + "_proj", x).reshape(kv_heads, width) for name in ("k", "v") + ) + for name, value in ( + ("attention_q_projected", q), + ("attention_k_projected", k), + ("attention_v_projected", v), + ): + self.observe(position, layer, name, value) + q, k = ( + ref.rms_norm( + t, + self.weights.tensor(base + name + "_norm.weight"), + c["rms_norm_eps"], + weight_offset=1, + ) + for name, t in (("q", q), ("k", k)) + ) + rope = c["rope_parameters"] + self.observe(position, layer, "attention_q_norm", q) + self.observe(position, layer, "attention_k_norm", k) + rotary_dim = int( + width * rope.get("partial_rotary_factor", c.get("partial_rotary_factor", 1)) + ) + q, k = (ref.rope(t, position, rotary_dim, rope["rope_theta"]) for t in (q, k)) + self.observe(position, layer, "attention_q_rope", q) + self.observe(position, layer, "attention_k_rope", k) + + def encode(value, scale): + if self.precision["kv_fp8"]: + return ref.fp8_encode(value / scale) + return (ref.bf16(value).view("> 16).astype(" self.config["max_position_embeddings"] or any( + type(t) is not int or t < 0 or t >= self.config["vocab_size"] for t in tokens + ): + raise DiagnosticError("snapshot tokens exceed reference domain") + state = {} + for layer, kind in enumerate(self.config["layer_types"]): + names = ( + ("gdn", "conv") if kind == "linear_attention" else ("keys", "values", "kv_scales") + ) + state[layer] = {name: arrays[f"layer.{layer:03d}.{name}"] for name in names} + for name, value in state[layer].items(): + expected = self.state[layer][name] + if name in {"keys", "values"}: + encoding = document["components"][f"layer.{layer:03d}.{name}"]["dtype"] + if encoding != self.precision["kv_encoding"]: + raise DiagnosticError("snapshot KV encoding differs from the reference") + if not np.isfinite(ref.values(value.tobytes(), encoding)).all(): + raise DiagnosticError("snapshot contains nonfinite KV state") + shape = ( + (len(tokens), *expected.shape[1:]) + if name in {"keys", "values"} + else expected.shape + ) + if ( + value.dtype != expected.dtype + or value.shape != shape + or not np.isfinite(value).all() + ): + raise DiagnosticError("snapshot state geometry or precision changed") + if kind == "full_attention" and not np.array_equal( + state[layer]["kv_scales"], self.kv_scales[str(layer)] + ): + raise DiagnosticError("snapshot KV quantizer scales changed") + self.tokens, self.state = tokens, state + + def close(self): + self.weights.close() diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_native_reference.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_native_reference.py new file mode 100644 index 0000000..417c5bf --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_native_reference.py @@ -0,0 +1,198 @@ +"""CPU-only projection of an authenticated serial baseline onto a shorter replay. + +This does not execute a model, qualify a native implementation, or select a +cache entry. The caller must bind native configuration and artifacts separately. +Every retained frame is independently copied and its actual bytes authenticated. +""" + +from pathlib import Path + +from qwen_r9700_lab.conformance_boundaries import validate_domain +from qwen_r9700_lab.conformance_replay import SCHEDULE_SCHEMA, observation_domain, scheduled_inputs +from qwen_r9700_lab.conformance_state import archive_frame, read_frame +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + private_json, + seal, + write_private, +) + + +def _require(condition, message): + if not condition: + raise DiagnosticError(message) + + +def compatible_serial_plans(source, requested): + for plan in (source, requested): + authenticate(plan) + _require("accepted_widths" not in plan, "reference projection requires serial M1 plans") + for field in ("prefix", "forced_tokens"): + values = plan.get(field) + _require( + isinstance(values, list) + and bool(values) + and all(type(v) is int and 0 <= v < 2**31 for v in values), + "invalid serial token domain", + ) + positions, _ = observation_domain(plan) + _require( + positions + and all(type(p) is int for p in positions) + and positions == sorted(set(positions)) + and positions[0] >= 0 + and positions[-1] < len(plan["prefix"]) + len(plan["forced_tokens"]) - 1, + "invalid serial observation domain", + ) + variable = {"sha256", "forced_tokens", "observation_positions"} + _require( + {k: v for k, v in source.items() if k not in variable} + == {k: v for k, v in requested.items() if k not in variable}, + "reference contract, artifact or initial prefix changed", + ) + forced = requested["forced_tokens"] + _require( + source["forced_tokens"][: len(forced)] == forced, + "requested continuation is not the recorded serial prefix", + ) + + +def _schedule(plan, capture): + schedule = private_json(capture / "schedule.json") + authenticate(schedule) + _require( + schedule.get("schema") == SCHEDULE_SCHEMA + and schedule.get("plan") == plan["sha256"] + and schedule.get("contract") == plan["contract"] + and schedule.get("initial_state") == "independent_zero_state" + and schedule.get("published_to_session") is False, + "serial baseline provenance is incomplete or incompatible", + ) + expected = list(scheduled_inputs(plan)) + frames = schedule.get("frames", []) + _require(len(frames) == len(expected), "serial baseline schedule is incomplete") + for actual, wanted in zip(frames, expected, strict=True): + _require( + all(actual.get(k) == v for k, v in wanted.items()), + "serial baseline did not follow its declared schedule", + ) + return schedule + + +def project_serial_reference( + source_plan, requested_plan, source_capture: Path, output: Path, *, reflink: bool = False +): + """Retain exactly the requested prefix/domain, with explicit source lineage. + + A failed copy leaves diagnostic partial output without a completion receipt. + It never alters the source. No physical allocation IDs are compared. Reflinks + retain independent inodes and verified bytes; unsupported storage fails + rather than silently copying the full baseline into a limited filesystem. + """ + compatible_serial_plans(source_plan, requested_plan) + source_capture, output = Path(source_capture), Path(output) + schedule = _schedule(source_plan, source_capture) + boundaries = private_json(source_capture / "boundaries/boundaries.json") + authenticate(boundaries) + positions, _ = observation_domain(source_plan) + requested_positions, input_digests = observation_domain(requested_plan) + stages = boundaries.get("layer_stages") + validate_domain(positions, boundaries.get("layers"), stages) + _require( + boundaries.get("schema") == "urn:qwen:boundary-schedule:v2" + and boundaries.get("positions") == positions, + "serial baseline boundary provenance changed", + ) + expected_names = [ + f"p{p:09d}-l{layer:03d}-{stage}" + for p in positions + for layer, layer_stages in enumerate(stages) + for stage in layer_stages + ] + _require( + [f["name"] for f in boundaries["frames"]] == expected_names + and set(requested_positions) <= set(positions), + "serial baseline is missing required boundary observations", + ) + output.mkdir(mode=0o700) + source_frames = {f["name"]: f for f in schedule["frames"]} + projected_frames = [] + for wanted in scheduled_inputs(requested_plan): + entry = source_frames[wanted["name"]] + _require(all(entry[k] == v for k, v in wanted.items()), "unaligned serial prefix frame") + original = read_frame(source_capture / entry["name"]) + _require( + original["sha256"] == entry["sha256"] + and original["contract"] == source_plan["contract"] + and original["coverage"] == schedule["coverage"] + and all(original[k] == v for k, v in wanted.items() if k != "name"), + "serial baseline frame identity changed", + ) + archive_frame(source_capture / entry["name"], output / entry["name"], reflink=reflink) + projected_frames.append(entry) + boundary_output = output / "boundaries" + boundary_output.mkdir(mode=0o700) + by_name = {f["name"]: f for f in boundaries["frames"]} + projected_boundaries = [] + for position in requested_positions: + for layer, layer_stages in enumerate(stages): + for stage in layer_stages: + name = f"p{position:09d}-l{layer:03d}-{stage}" + entry = by_name[name] + path = source_capture / "boundaries" / name + frame = read_frame(path) + _require( + frame["sha256"] == entry["sha256"] + and frame["contract"] == source_plan["contract"] + and frame["input_digest"] == input_digests[position] + and frame["consumed"] == position + 1 + and frame["pending"] is None + and frame["phase"] == "operator" + and frame["logical"] == {"layer": layer, "stage": stage} + and frame["coverage"] == ["value"], + "boundary identity changed", + ) + archive_frame(path, boundary_output / name, reflink=reflink) + projected_boundaries.append(entry) + lineage = { + "source_plan": source_plan["sha256"], + "requested_plan": requested_plan["sha256"], + "source_schedule": schedule["sha256"], + "source_boundaries": boundaries["sha256"], + } + projected_schedule = seal( + { + **{k: v for k, v in schedule.items() if k != "sha256"}, + "plan": requested_plan["sha256"], + "frames": projected_frames, + "reference_projection": lineage, + } + ) + projected_boundary_schedule = seal( + { + **{k: v for k, v in boundaries.items() if k != "sha256"}, + "positions": requested_positions, + "frames": projected_boundaries, + "reference_projection": lineage, + } + ) + write_private(output / "schedule.json", projected_schedule) + write_private(boundary_output / "boundaries.json", projected_boundary_schedule) + write_private(output / "plan.json", requested_plan) + result = seal( + { + "schema": "urn:qwen:serial-reference-projection:v1", + **lineage, + "schedule": projected_schedule["sha256"], + "boundaries": projected_boundary_schedule["sha256"], + "states": len(projected_frames), + "observations": len(projected_boundaries), + "storage": ( + "independent verified reflinks" if reflink else "independent verified copies" + ), + "native_equivalence": "UNPROVED", + } + ) + write_private(output / "reference-projection.json", result) + return result diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_obligations.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_obligations.py new file mode 100644 index 0000000..00600ca --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_obligations.py @@ -0,0 +1,338 @@ +"""Explicit proof obligations and implementation coverage, not a certificate. + +Each row names an actual instrumentation/checking path and its remaining gap. +An observed execution or a helper theorem cannot discharge a backend theorem. +The inventory is scoped to this text-only, single-GPU Radiance deployment. +""" + +from pathlib import Path + +from qwen_r9700_lab.conformance_artifacts import file_identity +from qwen_r9700_lab.conformance_reference import reference_contract +from qwen_r9700_lab.diagnostic_contract import seal + +# id, obligation, available implementation, regression suite, unresolved link. +COMPONENTS = ( + ( + "reference", + "R_C implements the declared finite-precision model on every admitted input", + "conformance_reference.py", + "test_conformance_reference.py", + ( + "NumPy transcendental/rounding behavior and the stock model " + "implementation are not proved equivalent" + ), + ), + ( + "prompt", + "Tokens_B(u) = Tokens_R(u), including positions, masks and template", + "conformance_replay.py", + "test_conformance_replay.py", + ( + "Replay begins with explicit token IDs; live tokenizer, template and " + "request rendering are outside it" + ), + ), + ( + "weights", + "Decode_B(Q(W), i) = Decode_R(Q(W), i) for every admitted packed code/index", + "conformance_model.py", + "test_conformance_reference.py", + ( + "Nibble helpers are checked; native permutation, E8M0 scale application " + "and all native shapes need comparison" + ), + ), + ( + "activations", + "Quant_B(x) = Quant_C(x); no undeclared activation quantizer", + "conformance_reference.py", + "test_conformance_profiles.py", + ( + "FP8 activations intentionally change the weight-only reference; actual " + "dispatch needs validation" + ), + ), + ( + "normalization", + "Norm_B(x) = Norm_R(x), including reductions and rounding points", + "conformance_boundaries.py", + "test_conformance_boundaries.py", + ( + "Common native module outputs are captured; fused internal operations are" + " not independently qualified" + ), + ), + ( + "linear", + "Linear_B(x,Q(W)) = Linear_R(x,Q(W))", + "conformance_instrumentation.py", + "test_conformance_instrumentation.py", + ( + "Python calls and tensors are observable; HIP/ISA arithmetic and all " + "dispatch regimes remain unproved" + ), + ), + ( + "rope", + "RoPE_B(x,p) = RoPE_R(x,p) for every admitted position", + "conformance_reference.py", + "test_conformance_reference.py", + "Native rotation/trigonometric implementations need same-input operator replay", + ), + ( + "attention", + "Attention_B(q,K,V,mask) = Attention_R(q,K,V,mask)", + "conformance_radiance.py", + "test_conformance_radiance.py", + ( + "Native R4D entrypoints can be traced; complete causality, selector, " + "reduction and scratch-lifetime proofs are missing" + ), + ), + ( + "kv", + "alpha(KV_B)[p] = KV_R[p], including declared dtype/scales", + "conformance_state.py", + "test_conformance_profiles.py", + ( + "Byte comparison rejects missing values; BF16 native extraction and GPU " + "write completion remain unvalidated" + ), + ), + ( + "convolution", + "(y,H')_B = (y,H')_R; rejected suffix cannot change H'", + "conformance_radiance.py", + "test_conformance_radiance.py", + ( + "Selected history offsets are checked; the real temporal bank mapping and" + " convolution kernel need native replay" + ), + ), + ( + "gdn", + "(y,S')_B = (y,S')_R, including every committed recurrent-state element", + "conformance_model.py", + "test_conformance_replay.py", + ( + "Independent recurrence and glue captures exist; native " + "prefill/verify/rollback equivalence is not proved" + ), + ), + ( + "prefill", + "alpha(Prefill_B(t_0..t_n)) = Fold(R_C, t_0..t_n)", + "conformance_replay.py", + "test_conformance_boundaries.py", + ( + "Every materialized token can be observed; native intermediate tensors, " + "all chunk sizes and actual schedules need qualification" + ), + ), + ( + "head", + "argmax_full(z_B) = argmax_full(z_R); excluded candidates cannot win", + "conformance_invariants.py", + "test_conformance_control.py", + ( + "Interval certificate assumes sound bounds; INT2 shortlist exact " + "reranking alone is insufficient" + ), + ), + ( + "speculation", + "Target_B(P,d_=j has no effect", + "conformance_radiance.py", + "test_conformance_replay.py", + ( + "Forced D7 widths check selected state; native acceptance algorithm, " + "proposer and all dynamic widths need separate evidence" + ), + ), + ( + "commit", + "alpha(Commit_B(P,d,k)) = R_C on exactly the processed accepted prefix", + "conformance_gate.py", + "test_conformance_proofs.py", + ( + "Eight-slot copy/pending-token helpers have scoped proofs; these are not " + "proofs of Radiance's implementation" + ), + ), + ( + "ownership", + "No writable alias of shared state; unrelated logical sessions are unchanged", + "conformance_control.py", + "test_conformance_control.py", + ( + "Control trace invariants are checked; producer mapping, device races and" + " physical ownership need native validation" + ), + ), + ( + "scheduler", + "Interleavings preserve each session's trace and state; stale epochs cannot publish", + "conformance_control.py", + "test_radiance_priority.py", + ( + "CPU scheduling cases exist; production concurrency and priority " + "preemption have no native conformance proof" + ), + ), + ( + "snapshot", + "Load_C(Save_C(S)) = S and every continued suffix has equal observables", + "conformance_lifecycle.py", + "test_conformance_lifecycle.py", + ( + "Real storage transport plus diagnostic frames is tested; the actual " + "native snapshot mapping and failure schedules are unqualified" + ), + ), + ( + "sampling", + "Greedy_B(z)=Greedy_R(z); sampled modes need identical conditional distributions", + "conformance_invariants.py", + "test_conformance_control.py", + ( + "Replay forces tokens; it does not certify production penalties, RNG, " + "rejection sampling, grammar or stopping" + ), + ), + ( + "protocol", + "Tokens, reasoning/content, tool events, finish status and pending state match R_C", + "conformance_session.py", + "test_conformance_session.py", + ( + "Prototype checks tokens/stops; live tool parser, streaming transport and" + " Pi publication are outside the gate" + ), + ), + ( + "gate", + ( + "Publish => equal output and next state, complete coverage, completed " + "writes and current epoch" + ), + "conformance_invariants.py", + "test_conformance_session.py", + ( + "Prototype gate is tested and helper expressions proved; the production " + "backend is not connected to it" + ), + ), + ( + "dispatch", + "Every executed implementation satisfies C and its declared preconditions", + "conformance_dispatch.py", + "test_conformance_dispatch.py", + ( + "Reviewed library exports/explicit aliases can be recorded; hidden C++ " + "launches and graph replay are outside this observer" + ), + ), + ( + "graphs", + ( + "Graph replay refines the serial transition with no stale pointers, races" + " or lifetime violations" + ), + "conformance_instrumentation.py", + "test_conformance_instrumentation.py", + ( + "Tensor observation is serialized/eager; it can hide races in the " + "production asynchronous schedule" + ), + ), + ( + "compiler", + "Compiled(K) refines the proved source under declared ISA semantics", + "conformance_artifacts.py", + "test_conformance_artifacts.py", + ( + "Mapped library hashes are not actual HSACO dispatch identity or " + "compiler/ISA translation validation" + ), + ), + ( + "machine", + "Execution follows the declared memory, arithmetic and isolation model", + "conformance_artifacts.py", + "test_conformance_artifacts.py", + ( + "GPU hardware, driver, firmware, host runtime, reference isolation and " + "fault-free execution remain assumptions" + ), + ), +) + + +def proof_obligations(profile="weight-only-bf16"): + package = Path(__file__).parent + root = package.parent.parent + rows = [] + for identifier, formula, implementation, regression, gap in COMPONENTS: + paths = (package / implementation, root / "tests" / regression) + rows.append( + { + "id": identifier, + "formula": formula, + "backend_status": "UNPROVED", + "available_checks": { + str(path.relative_to(root)): file_identity(path) + if path.is_file() + else "UNAVAILABLE" + for path in paths + }, + "remaining_obligation": gap, + } + ) + return seal( + { + "schema": "urn:qwen:backend-proof-obligations:v1", + "reference": reference_contract(profile), + "domain": ( + "declared text-only dense Qwen3.5 architecture, one GPU, greedy " + "reference; admitted finite inputs and reachable states" + ), + "logical_state": [ + "ordered KV and quantizers", + "GDN state", + "convolution history", + "consumed positions", + "emitted but pending token", + "ownership and versions", + "sampler/stop/parser state where admitted", + ], + "theorem": ( + "forall (S,P,u) in D: alpha(P)=S => alpha(B_C(P,u).state)=R_C(S,u).state " + "AND Obs(B_C(P,u))=Obs(R_C(S,u))" + ), + "composition": { + "initialization": "Init_B and Init_R satisfy alpha(P_0)=S_0", + "step": "Every admitted transition preserves alpha and exactly matches observables", + "frame": "A transition cannot modify another session or a shared immutable prefix", + "closure": "Every result remains within the next transition's proved preconditions", + "induction": ( + "Initialization + step + frame + closure imply equal finite published traces" + ), + "availability": ( + "Equal published prefixes alone do not prove termination or service " + "availability" + ), + }, + "obligations": rows, + "undischarged": [row["id"] for row in rows], + "universal_equivalence": "UNPROVED", + "stock_quantized_equivalence": "UNPROVED", + "instrumentation_complete": False, + "production_gate_installed": False, + "evidence_rule": ( + "TESTED and RUNTIME-CHECKED are scoped observations, never universal " + "proof; helper proofs do not discharge native obligations" + ), + "gpu_used": False, + } + ) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_observer.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_observer.py new file mode 100644 index 0000000..30ba39a --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_observer.py @@ -0,0 +1,415 @@ +"""Source-bound observers loaded only in an owned qualification subprocess. + +Metadata mode does not copy tensors or synchronize the GPU. Head-audit mode +deliberately does both and is reported separately. Neither mode is installed in +the live backend. Host entry/return records do not attest individual device ISA +instructions or prove that an asynchronous launch has completed. +""" + +from __future__ import annotations + +import functools +import hashlib +import importlib.abc +import importlib.machinery +import inspect +import json +import os +import sys +import threading +import time +from pathlib import Path + +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, digest, private_json + + +class Events: + def __init__(self, root, execution): + self.root, self.execution = Path(root), execution + self.pid, self.sequence, self.previous = os.getpid(), 0, "0" * 64 + self.lock = threading.Lock() + os.register_at_fork(after_in_child=self._after_fork) + + def _after_fork(self): + self.lock = threading.Lock() + + def emit(self, event, **details): + with self.lock: + pid = os.getpid() + if pid != self.pid: + self.pid, self.sequence, self.previous = pid, 0, "0" * 64 + row = { + "execution": self.execution, + "pid": pid, + "sequence": self.sequence, + "previous": self.previous, + "monotonic_ns": time.monotonic_ns(), + "event": event, + "details": details, + } + row["sha256"] = digest(row) + fd = os.open( + self.root / f"events-{self.execution}-{pid}.jsonl", + os.O_WRONLY | os.O_CREAT | os.O_APPEND, + 0o600, + ) + with os.fdopen(fd, "a") as output: + output.write(json.dumps(row, separators=(",", ":")) + "\n") + self.sequence += 1 + self.previous = row["sha256"] + + +def read_events(root, execution): + executions = {execution} if isinstance(execution, str) else set(execution) + all_rows = [] + files = sorted(Path(root).glob("events-*.jsonl")) + if not files: + raise DiagnosticError("no native path observations were produced") + for path in files: + previous, sequence = "0" * 64, 0 + with path.open() as stream: + for line in stream: + try: + row = json.loads(line) + hashed = {k: v for k, v in row.items() if k != "sha256"} + valid = ( + row["sha256"] == digest(hashed) + and row["previous"] == previous + and row["sequence"] == sequence + and row["execution"] in executions + and path.name == f"events-{row['execution']}-{row['pid']}.jsonl" + ) + except (ValueError, KeyError, TypeError) as error: + raise DiagnosticError("incomplete native event evidence") from error + if not valid: + raise DiagnosticError("native event identity or ordering changed") + previous, sequence = row["sha256"], sequence + 1 + all_rows.append(row) + return sorted(all_rows, key=lambda r: (r["monotonic_ns"], r["pid"], r["sequence"])) + + +def wrap(owner, name, events, event, *, details=None, after=None): + original = getattr(owner, name, None) + if not callable(original): + raise DiagnosticError("pinned native observer entry point is missing: " + event) + + @functools.wraps(original) + def observed(*args, **kwargs): + extra = details(args, kwargs) if details else {} + events.emit(event + ".enter", **extra) + try: + result = original(*args, **kwargs) + except BaseException as error: + events.emit(event + ".error", error_type=type(error).__name__) + raise + more = after(args, kwargs, result) if after else {} + events.emit(event + ".return", **extra, **more) + return result + + setattr(owner, name, observed) + + +def instrument_qwen_parser(cls, events): + # The deployed ParserManager retains compatibility adapters for structural + # tags. Those adapters bypass parse_delta and call the extraction methods. + # Observe the actual inherited methods as well as the direct engine API. + for method, event in { + "parse_delta": "parser.delta", + "extract_tool_calls_streaming": "parser.tool_stream", + "extract_tool_calls_from_content": "parser.tool_nonstream", + "extract_reasoning_streaming": "parser.reasoning_stream", + "extract_reasoning": "parser.reasoning_nonstream", + "finish_streaming": "parser.finish", + }.items(): + wrap(cls, method, events, event) + + +def instrument(module, settings, events): + name = module.__name__ + expected = settings["observer_sources"].get(name) + path = Path(inspect.getfile(module)).resolve() + if expected != hashlib.sha256(path.read_bytes()).hexdigest(): + raise DiagnosticError("native observation source differs: " + name) + events.emit("source.bound", module=name, source_sha256=expected) + if name == "vllm.v1.worker.gpu.model_runner": + cls = module.GPUModelRunner + + def prepared(args, kwargs, result): + args[0]._qwen_campaign_batch = result + return { + "num_reqs": int(result.num_reqs), + "num_tokens": int(result.num_tokens), + "draft_tokens": int(result.num_draft_tokens), + } + + wrap(cls, "prepare_inputs", events, "runner.prepare", after=prepared) + + def committed(args, kwargs, result): + capture_at_boundary(args[0], settings, events) + return {} + + wrap(cls, "postprocess_sampled", events, "runner.commit", after=committed) + # This records the real replay call, not merely a graph-enabled flag. + import torch + + wrap(torch.cuda.CUDAGraph, "replay", events, "graph.replay") + elif name == "qwen_radiance_fair_scheduler": + wrap( + module.FairScheduler, + "answer_priority", + events, + "priority.apply", + details=lambda a, k: { + key: a[1].get(key) for key in ("chat_id", "sequence", "priority", "active") + }, + ) + wrap( + module.RequestPhases, + "finish", + events, + "response.finish", + details=lambda a, k: { + "chat_id": (a[1].kv_transfer_params or {}).get("qwen_chat", {}).get("id"), + }, + ) + wrap( + module.FairScheduler, + "schedule", + events, + "scheduler.step", + after=lambda a, k, result: { + "scheduled": { + hashlib.sha256(str(r).encode()).hexdigest(): int(n) + for r, n in result.num_scheduled_tokens.items() + }, + "active": a[0].banks.active, + }, + ) + wrap(module.FairScheduler, "_retire_bank", events, "bank.retire") + wrap( + module.WorkerBanks, + "before", + events, + "bank.activate", + after=lambda a, k, result: { + "active": a[0].active, + "ram_banks": len(a[0].images), + }, + ) + elif name == "qwen_radiance_chat_tier": + for method, event in { + "_load": "snapshot.load", + "_store": "snapshot.store", + "_flush_record": "snapshot.flush", + "shutdown": "snapshot.shutdown", + }.items(): + wrap(module.ChatFileSystemTierManager, method, events, event) + elif name == "qwen_radiance_cache": + for method, event in {"publish": "snapshot.publish", "collect": "snapshot.collect"}.items(): + wrap(module.ChatStore, method, events, event) + original = module.atomic_write + + def atomic(path, content): + path = Path(path) + barrier = Path(settings["root"]) / "interrupt-write.arm" + if path.suffix == ".qkv" and barrier.exists(): + # Leave an actual partial temporary payload beside the old + # complete head. Only the owned child can reach this branch. + temporary = path.parent / ".pending-conformance-interrupted" + fd = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(fd, "wb") as stream: + stream.write(content[: max(1, len(content) // 2)]) + stream.flush() + os.fsync(stream.fileno()) + events.emit("snapshot.partial_write", payload_bytes=len(content)) + # The controller kills THIS process group after observing the + # receipt. A deadline prevents an abandoned controller hanging. + deadline = time.monotonic() + 30 + while barrier.exists() and time.monotonic() < deadline: + time.sleep(0.05) + raise DiagnosticError("intentional interrupted snapshot write") + return original(path, content) + + module.atomic_write = atomic + elif name == "radiance_verifyhead": + + def head_details(args, kwargs): + return {"fast": bool(getattr(args[0], "_radiance_fast_ok", False))} + + def head_after(args, kwargs, result): + if not settings.get("head_audit") or not getattr(args[0], "_radiance_fast_ok", False): + return {} + import torch + + if torch.cuda.is_current_stream_capturing(): + return {"head_audit": "capture_not_compared"} + state, lm_head, hidden = args[:3] + bias = args[3] if len(args) > 3 else kwargs.get("embedding_bias") + exact = state._radiance_exact_head(lm_head, hidden, bias) + fast = result.float().reshape(-1, result.shape[-1]) + exact = exact.float().reshape(-1, exact.shape[-1]) + if fast.shape != exact.shape or not torch.isfinite(exact).all(): + raise DiagnosticError("invalid exact head observations") + if settings.get("head_fault"): + # Fault only the diagnostic comparison input, on the actual + # device. The production result object is returned unchanged. + fast = fast.clone() + fast.masked_fill_(exact == exact.amax(-1, keepdim=True), -float("inf")) + events.emit( + "head.fault_applied", fault="omitted_exact_maxima", device=str(fast.device) + ) + finite = torch.isfinite(fast) + if torch.isnan(fast).any() or torch.isposinf(fast).any() or not finite.any(-1).all(): + raise DiagnosticError("invalid fast head observations") + omitted = exact.amax(-1) > exact.masked_fill(~finite, -float("inf")).amax(-1) + report = { + "head_audit": "compared", + "rows": int(fast.shape[0]), + "omitted_winners": int(omitted.sum()), + "different_retained_logits": int(((fast != exact) & finite).sum()), + "different_argmax": int((fast.argmax(-1) != exact.argmax(-1)).sum()), + } + if ( + report["omitted_winners"] + or report["different_retained_logits"] + or report["different_argmax"] + ): + # Preserve actual counterexample tensors, with no text decoding. + filename = ( + Path(settings["root"]) + / f"head-discrepancy-{os.getpid()}-{time.monotonic_ns()}.pt" + ) + fd = os.open(filename, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(fd, "wb") as stream: + torch.save( + { + "hidden": hidden.detach().cpu(), + "fast": fast.detach().cpu(), + "exact": exact.detach().cpu(), + }, + stream, + ) + report["counterexample"] = filename.name + return report + + wrap( + module, + "_apply_head_gated", + events, + "head.apply", + details=head_details, + after=head_after, + ) + elif name == "vllm.parser.qwen3": + instrument_qwen_parser(module.Qwen3Parser, events) + elif name in { + "vllm.parser.engine.parser_engine", + "vllm.parser.engine.adapters", + "vllm.parser.parser_manager", + }: + # Bind the inherited implementation and adapter selection as well as + # Qwen's grammar. Invocation observations are on the concrete class. + pass + else: + raise DiagnosticError("unknown observer module") + + +def capture_at_boundary(runner, settings, events): + """Actual synchronous post-commit capture, before the next scheduler step. + + Prefix identity comes from the controller's synthetic input, not from the + producer being checked. This diagnostic is separate from async/race runs. + """ + request_path = Path(settings["root"]) / "capture-request.json" + if not request_path.exists(): + return + from types import SimpleNamespace + + from qwen_r9700_lab.conformance_radiance import capture_committed_state + + request = private_json(request_path) + output = Path(settings["root"]) / request["name"] + if output.exists(): + return + if output.parent != Path(settings["root"]) or not output.name.startswith("state-"): + raise DiagnosticError("unsafe qualification capture destination") + if runner.vllm_config.scheduler_config.async_scheduling: + raise DiagnosticError("synchronous state capture cannot qualify async execution") + batch = runner._qwen_campaign_batch + if batch.num_reqs != 1: + raise DiagnosticError("state capture requires one dispatched request") + index = int(batch.idx_mapping_np[0]) + consumed = int(runner.req_states.num_computed_tokens.gpu[index].item()) + if consumed < request["consumed"]: + return + if consumed != request["consumed"]: + raise DiagnosticError("native state capture missed its exact token boundary") + pending = int(runner.req_states.last_sampled_tokens[index, 0].item()) + requests = [rid for rid, idx in runner.req_states.req_id_to_index.items() if idx == index] + if len(requests) != 1: + raise DiagnosticError("native capture has ambiguous request ownership") + capture_committed_state( + SimpleNamespace(model_runner=runner), + output_path=str(output), + request_id=requests[0], + expected={ + "consumed": consumed, + "pending": pending, + "input_digest": request["input_digest"], + }, + plan_path=request["plan"], + binding_path=request["binding"], + quiescent=True, + ) + events.emit("state.captured", name=output.name, consumed=consumed) + + +class ObserverFinder(importlib.abc.MetaPathFinder): + def __init__(self, settings, events): + self.settings, self.events = settings, events + + def find_spec(self, fullname, path=None, target=None): + if fullname not in self.settings["observer_sources"]: + return None + spec = importlib.machinery.PathFinder.find_spec(fullname, path) + if spec is None or spec.loader is None: + raise DiagnosticError("pinned observation module was not found") + original = spec.loader + settings, events = self.settings, self.events + + class Loader(importlib.abc.Loader): + def create_module(self, inner): + return original.create_module(inner) + + def exec_module(self, module): + if ( + spec.origin is None + or hashlib.sha256(Path(spec.origin).read_bytes()).hexdigest() + != settings["observer_sources"][fullname] + ): + raise DiagnosticError( + "native observer source differs before import: " + fullname + ) + original.exec_module(module) + instrument(module, settings, events) + + spec.loader = Loader() + return spec + + +def install_from_environment(): + path = os.environ.get("QWEN_CONFORMANCE_SERVER_SETTINGS") + if not path or os.environ.get("QWEN_CONFORMANCE_GPU") != "1": + raise DiagnosticError("owned native observer is not armed") + settings = private_json(Path(path)) + root = Path(settings["root"]) + if not root.is_dir() or root.stat().st_mode & 0o077: + raise DiagnosticError("qualification root is not private") + events = Events(root, settings["execution"]) + for module in settings["observer_sources"]: + if module in sys.modules: + raise DiagnosticError("native module loaded before observer binding") + sys.meta_path.insert(0, ObserverFinder(settings, events)) + events.emit( + "observer.ready", scope="host_entry_return", head_audit=settings.get("head_audit", False) + ) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_parser.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_parser.py new file mode 100644 index 0000000..b4394ab --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_parser.py @@ -0,0 +1,118 @@ +"""Actual Qwen parser over synthetic token streams, including tool boundaries.""" + +from __future__ import annotations + +import hashlib +import inspect +from pathlib import Path + +from qwen_r9700_lab.conformance_transport import Completion, ProtocolError +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, write_private + + +def qualify_parser(checkpoint, expected_sha256, output, *, parser_config=None): + from transformers import AutoTokenizer + from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest + from vllm.parser.qwen3 import Qwen3Parser + + source = Path(inspect.getfile(Qwen3Parser)) + if hashlib.sha256(source.read_bytes()).hexdigest() != expected_sha256: + raise DiagnosticError("native parser binding changed") + parser_class = Qwen3Parser + if parser_config is not None: + from vllm.parser.parser_manager import ParserManager + + parser_class = ParserManager.get_parser(**parser_config) + if parser_class is None: + raise DiagnosticError("configured serving parser is unavailable") + tokenizer = AutoTokenizer.from_pretrained(checkpoint, local_files_only=True) + tool = { + "type": "function", + "function": { + "name": "record", + "parameters": { + "type": "object", + "properties": {"text": {"type": "string"}}, + "required": ["text"], + }, + }, + } + request = ChatCompletionRequest( + model="fixture", + messages=[{"role": "user", "content": "Record."}], + tools=[tool], + tool_choice="auto", + stream=True, + ) + xml = ( + "\n\ncafé λ 🙂\n" + "\n" + ) + fixtures = { + "valid_tool": ("Plan.\n\n\n" + xml, True), + "marker_in_thinking": ('The token "" is syntax.\n\n\n' + xml, True), + "answer_only": ("Plan.\n\n\nThe value is 37.", False), + "colon_only": ("Plan.\n\n\nThe next step is:", False), + "unfinished_name": ("Plan.\n\n\n\n 4096 + # No arbitrary error text, request contents or authorization headers + # enter the diagnostic receipt. + raise + finally: + receipt["finished_ns"] = time.monotonic_ns() + receipt["seconds"] = (receipt["finished_ns"] - receipt["started_ns"]) / 1e9 + with self.lock: + self.pending -= 1 + target = ( + self.confirmations if receipt.get("status") == "confirmed" else self.failures + ) + target.append(receipt) + write_private(self.root / (name + ".result.json"), receipt) + + def close(self): + with self.lock: + self.closed, self.active = True, False + self.stop.set() + if self.heart: + self.heart.join() + self.pool.shutdown(wait=True) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_proofs.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_proofs.py new file mode 100644 index 0000000..9595543 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_proofs.py @@ -0,0 +1,397 @@ +"""Small, scoped SMT obligations over actual prototype helper expressions. + +This does not prove Radiance, the checker implementation, Python or GPU code. +The solver, Python/Z3 symbolic-expression translation and machine are trusted. +Every counterexample query, implementation hash and result is preserved. +""" + +from __future__ import annotations + +import hashlib +import inspect +from pathlib import Path + +from qwen_r9700_lab.conformance_gate import copy_accepted_prefix, publication_allowed +from qwen_r9700_lab.diagnostic_contract import integer, remaining_materialized, seal, write_private + + +def check_obligation( + output_root, name, counterexample, *, expected, assumptions=(), timeout_ms=10_000 +): + """Preserve the query and witness; inconsistent domains never prove a claim.""" + import z3 + + integer(timeout_ms, minimum=1) + domain = z3.Solver() + domain.set(timeout=timeout_ms) + domain.add(*assumptions) + domain_query = domain.to_smt2() + (output_root / (name + ".domain.smt2")).write_text(domain_query) + domain_result = domain.check() + solver = z3.Solver() + solver.set(timeout=timeout_ms) + solver.add(*assumptions, counterexample) + query = solver.to_smt2() + (output_root / (name + ".smt2")).write_text(query) + actual = solver.check() + row = { + "name": name, + "expected": str(expected), + "result": str(actual), + "domain_result": str(domain_result), + "domain_sha256": hashlib.sha256(domain_query.encode()).hexdigest(), + "query_sha256": hashlib.sha256(query.encode()).hexdigest(), + "status": ("PROVED" if expected == z3.unsat else "TESTED") + if actual == expected and domain_result == z3.sat + else "UNPROVED", + } + if domain_result != z3.sat: + row["domain_error"] = "inconsistent or unproved preconditions" + if actual == z3.unknown: + row["unknown_reason"] = solver.reason_unknown() + if actual == z3.sat: + witness = solver.model().sexpr() + (output_root / (name + ".counterexample.smt2")).write_text(witness + "\n") + row["counterexample_sha256"] = hashlib.sha256((witness + "\n").encode()).hexdigest() + return row + + +def run_obligations(output_root: Path, timeout_ms: int = 10_000): + import z3 + + from qwen_r9700_lab.conformance_invariants import ( + margin_preserved, + restore_matches, + separated_intervals, + snapshot_publishable, + transaction_ready, + unresolved_frontiers_separated, + version_matches, + writable_exclusively, + ) + from qwen_r9700_lab.conformance_radiance import convolution_offset, temporal_column + from qwen_r9700_lab.conformance_reference import ( + bf16_round_bits, + canonical_slot, + high_nibble, + low_nibble, + ) + + integer(timeout_ms, minimum=1) + output_root.mkdir(mode=0o700, parents=True, exist_ok=False) + rows = [] + + def check(name, counterexample, *, expected=z3.unsat, assumptions=()): + rows.append( + check_obligation( + output_root, + name, + counterexample, + expected=expected, + assumptions=assumptions, + timeout_ms=timeout_ms, + ) + ) + + out, state, identity, complete = z3.Bools("output_equal state_equal identity_equal complete") + # The implementation function executes on symbolic values: its Boolean + # expression is not manually copied into the verification condition. + allowed = publication_allowed(out, state, identity, complete) + check("gate_requires_state", z3.And(allowed, z3.Not(state))) + check("gate_requires_output", z3.And(allowed, z3.Not(out))) + check("gate_requires_identity", z3.And(allowed, z3.Not(identity))) + check("gate_requires_coverage", z3.And(allowed, z3.Not(complete))) + check("mutant_output_only_gate", z3.And(out, z3.Not(state)), expected=z3.sat) + + current = tuple(z3.BitVec(f"old_{i}", 32) for i in range(8)) + proposed = tuple(z3.BitVec(f"candidate_{i}", 32) for i in range(8)) + alternative = tuple(z3.BitVec(f"perturbed_{i}", 32) for i in range(8)) + for count in range(9): + committed = copy_accepted_prefix(current, proposed, count) + changed = copy_accepted_prefix(current, alternative, count) + check( + f"prefix_selection_{count}", + z3.Or(*[committed[i] != (proposed[i] if i < count else current[i]) for i in range(8)]), + ) + check( + f"rejected_suffix_noninterference_{count}", + z3.Or(*[committed[i] != changed[i] for i in range(8)]), + assumptions=[proposed[i] == alternative[i] for i in range(count)], + ) + check( + "mutant_commit_all", + z3.Or(*[proposed[i] != current[i] for i in range(1, 8)]), + expected=z3.sat, + ) + + before, old_pending, emitted, new_pending = z3.Ints("before old_pending emitted new_pending") + after = remaining_materialized(before, old_pending, emitted, new_pending) + check("pending_token_conservation", after + new_pending != before + old_pending + emitted) + check( + "materialized_nonnegative", + after < 0, + assumptions=[ + before >= 0, + old_pending >= 0, + old_pending <= 1, + emitted >= 1, + new_pending >= 0, + new_pending <= 1, + ], + ) + + fp = z3.Float32() + large, one, negative = (z3.FPVal(v, fp) for v in (16777216, 1, -16777216)) + rnd = z3.RNE() + left = z3.fpAdd(rnd, z3.fpAdd(rnd, large, one), negative) + right = z3.fpAdd(rnd, large, z3.fpAdd(rnd, one, negative)) + check("float32_reassociation_counterexample", z3.Not(z3.fpEQ(left, right)), expected=z3.sat) + + # Execute the actual NumPy bit expression symbolically. The independent + # oracle splits retained/discarded fields and applies the RNE rule. This + # proves the helper on its finite/Inf input domain, not GPU cast instructions. + bits = z3.BitVec("float32_bits", 32) + rounded = bf16_round_bits(bits) + high, low = z3.Extract(31, 16, bits), z3.Extract(15, 0, bits) + finite_or_inf = z3.Or((bits & 0x7F800000) != 0x7F800000, (bits & 0x007FFFFF) == 0) + increment = z3.If( + z3.Or(z3.UGT(low, 0x8000), z3.And(low == 0x8000, (high & 1) == 1)), + z3.BitVecVal(1, 16), + z3.BitVecVal(0, 16), + ) + independent = high + increment + check( + "bf16_rounding_matches_rne", + rounded != z3.Concat(independent, z3.BitVecVal(0, 16)), + assumptions=[finite_or_inf], + ) + check( + "bf16_rounding_clears_discarded_bits", (rounded & 0xFFFF) != 0, assumptions=[finite_or_inf] + ) + check( + "bf16_representable_values_unchanged", + rounded != bits, + assumptions=[finite_or_inf, low == 0], + ) + check( + "bf16_halfway_ties_even", + (rounded & 0x10000) != 0, + assumptions=[finite_or_inf, low == 0x8000], + ) + check( + "mutant_bf16_truncation", + (bits & 0xFFFF0000) != rounded, + expected=z3.sat, + assumptions=[finite_or_inf], + ) + + packed = z3.BitVec("packed_byte", 8) + check("mxfp4_nibble_roundtrip", ((high_nibble(packed) << 4) | low_nibble(packed)) != packed) + check("mxfp4_low_nibble_unsigned_range", z3.UGT(low_nibble(packed), 15)) + check("mxfp4_high_nibble_unsigned_range", z3.UGT(high_nibble(packed), 15)) + block, offset, size = z3.Ints("logical_block logical_offset logical_block_size") + slot = canonical_slot(block, offset, size) + domain = [block >= 0, size > 0, offset >= 0, offset < size] + check("canonical_slot_inverse_block", slot / size != block, assumptions=domain) + check("canonical_slot_inverse_offset", slot % size != offset, assumptions=domain) + running, accepted = z3.Ints("running_state_column accepted_count") + domain = [running >= 0, accepted >= 1, accepted <= 8] + check( + "accepted_state_column", + temporal_column(running, accepted) - running != convolution_offset(accepted), + assumptions=domain, + ) + check( + "accepted_conv_window_range", + z3.Or(convolution_offset(accepted) < 0, convolution_offset(accepted) > 7), + assumptions=domain, + ) + check("mutant_unshifted_conv_window", accepted > 1, expected=z3.sat, assumptions=domain) + + completed, cancelled, verified, durable = z3.Bools("completed cancelled verified durable") + base, epoch = z3.Ints("base_revision current_revision") + ready = transaction_ready(out, state, identity, complete, completed, cancelled, base, epoch) + for name, violation in ( + ("transaction_requires_completed_writes", z3.Not(completed)), + ("transaction_rejects_cancelled_work", cancelled), + ("transaction_rejects_stale_revision", base != epoch), + ("transaction_requires_full_equality", z3.Not(z3.And(out, state, identity, complete))), + ): + check(name, z3.And(ready, violation)) + check("mutant_unfenced_publication", z3.And(allowed, z3.Not(completed)), expected=z3.sat) + position, kv, gdn, conv = z3.Ints("position kv_position gdn_position conv_position") + check( + "all_state_versions_match", + z3.And( + version_matches(position, kv, gdn, conv), + z3.Or(kv != position, gdn != position, conv != position), + ), + ) + check("mutant_kv_only_version", z3.And(kv == position, gdn != position), expected=z3.sat) + refs, pins = z3.Ints("owner_references external_pins") + immutable = z3.Bool("immutable") + write = writable_exclusively(refs, pins, immutable) + check( + "shared_or_pinned_state_is_not_writable", + z3.And(write, z3.Or(refs != 1, pins != 0, immutable)), + ) + publish = snapshot_publishable(verified, durable, epoch, base) + check( + "snapshot_requires_verified_durable_current", + z3.And(publish, z3.Or(z3.Not(verified), z3.Not(durable), epoch != base)), + ) + check("mutant_publish_before_durability", z3.And(verified, z3.Not(durable)), expected=z3.sat) + ca, cb, ga, gb, sa, sb = z3.Ints( + "contract_a contract_b generation_a generation_b size_a size_b" + ) + check( + "restore_requires_same_contract_generation_position", + z3.And(restore_matches(ca, cb, ga, gb, sa, sb), z3.Or(ca != cb, ga != gb, sa != sb)), + ) + + # Conditional *real/rational* interval theorem. It does not infer these + # bounds from samples, and does not certify any GPU transcendental/kernel. + a, b, da, db, error = z3.Reals("winner runner_up winner_error runner_error certified_bound") + check( + "argmax_margin_bound", + a + da <= b + db, + assumptions=[ + margin_preserved(a - b, error), + da >= -error, + da <= error, + db >= -error, + db <= error, + ], + ) + lower, upper, actual_a, actual_b = z3.Reals( + "winner_lower other_upper actual_winner actual_other" + ) + check( + "full_vocabulary_interval_winner", + actual_a <= actual_b, + assumptions=[ + separated_intervals(lower, upper), + actual_a >= lower, + actual_b <= upper, + ], + ) + check("mutant_exact_shortlist_omits_winner", z3.And(a > b, actual_b > a), expected=z3.sat) + + # An arbitrary unresolved token must belong to one of the two covered + # partitions. This proves the comparison predicate under sound bounds; + # it does not derive those bounds or prove the native shortlist/partition. + cutoff, local_upper, global_upper, omitted_score = z3.Reals( + "topk_cutoff local_discard_upper global_discard_upper omitted_reference_score" + ) + local_unresolved, global_unresolved = z3.Bools("local_unresolved global_unresolved") + frontiers = unresolved_frontiers_separated( + cutoff, local_upper, global_upper, local_unresolved, global_unresolved + ) + covered = z3.Or( + z3.And(local_unresolved, omitted_score <= local_upper), + z3.And(global_unresolved, omitted_score <= global_upper), + ) + check( + "two_frontier_topk_bound", + omitted_score >= cutoff, + assumptions=[frontiers, covered], + ) + check( + "mutant_local_only_omits_global_candidate", + omitted_score > cutoff, + expected=z3.sat, + assumptions=[ + local_unresolved, + global_unresolved, + cutoff > local_upper, + omitted_score <= global_upper, + ], + ) + check( + "mutant_nonstrict_cutoff_accepts_omitted_tie", + omitted_score == cutoff, + expected=z3.sat, + assumptions=[cutoff >= global_upper, omitted_score <= global_upper], + ) + check( + "mutant_frontier_coverage_gap", + omitted_score > cutoff, + expected=z3.sat, + assumptions=[frontiers, z3.Not(covered)], + ) + + # A single induction step for a fail-closed published prefix. There is no + # availability claim when the gate refuses publication. Reference, isolation + # and event delivery remain explicit trusted preconditions. + reference_state, candidate_state = z3.BitVecs("reference_state candidate_state", 64) + reference_output, candidate_output = z3.BitVecs("reference_output candidate_output", 64) + step = transaction_ready( + reference_output == candidate_output, + reference_state == candidate_state, + identity, + complete, + completed, + cancelled, + base, + epoch, + ) + check( + "published_prefix_induction_step", + z3.And( + step, z3.Or(reference_state != candidate_state, reference_output != candidate_output) + ), + ) + + functions = { + f.__name__: hashlib.sha256(inspect.getsource(f).encode()).hexdigest() + for f in ( + copy_accepted_prefix, + publication_allowed, + remaining_materialized, + low_nibble, + high_nibble, + canonical_slot, + bf16_round_bits, + temporal_column, + convolution_offset, + transaction_ready, + version_matches, + writable_exclusively, + snapshot_publishable, + restore_matches, + margin_preserved, + separated_intervals, + unresolved_frontiers_separated, + ) + } + report = seal( + { + "schema": "urn:qwen:conformance-helper-obligations:v1", + "solver": z3.get_full_version(), + "timeout_ms": timeout_ms, + "functions_sha256": functions, + "cases": rows, + "all_expected_results": all(r["status"] != "UNPROVED" for r in rows), + "proved_scope": ( + "actual pure helper expressions: 8-slot prefix copy; byte nibble decoding; " + "unbounded-integer logical indexing and accepted state offsets; " + "control/publication predicates, finite/Inf FP32-to-BF16 RNE bit expression " + "and conditional rational interval/two-frontier cutoff inequalities; " + "not native GPU kernels" + ), + "trusted": [ + "Z3", + "Python execution and slicing", + "Z3 expression translation", + "hardware", + ], + "whole_gate_status": "TESTED", + "bounds_soundness": "ASSUMED; empirical errors cannot establish universal bounds", + "native_event_mapping": "UNPROVED", + "radiance_status": "UNPROVED", + "gpu_kernel_status": "UNPROVED", + "compiler_status": "ASSUMED", + } + ) + write_private(output_root / "results.json", report) + return report diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_protocol.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_protocol.py new file mode 100644 index 0000000..e85357d --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_protocol.py @@ -0,0 +1,108 @@ +"""Separate raw generation differences from output-parser presentation differences.""" + +from __future__ import annotations + +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, digest, seal, write_private + + +def _ids(value): + return ( + bool(value) + and isinstance(value, list) + and all(type(token) is int and token >= 0 for token in value) + ) + + +def _first_difference(a, b): + for index, (left, right) in enumerate(zip(a, b, strict=False)): + if left != right: + return index + return min(len(a), len(b)) if len(a) != len(b) else None + + +def compare_protocol_pair(streamed, nonstreamed, evidence, *, kind="stream_nonstream"): + """Preserve a bounded diagnostic, then enforce exact observed equivalence. + + Whitespace normalization is diagnostic only. A matching display is not enough + without complete raw IDs; a mismatch cannot be assigned to the parser unless + the prompt and generated IDs match. Tool IDs are transport-assigned identities + and may differ between independent requests; tool names/arguments may not. + """ + prompt = [result.get("prompt_token_ids") for result in (streamed, nonstreamed)] + output = [result.get("token_ids") for result in (streamed, nonstreamed)] + prompt_observed = all(_ids(ids) for ids in prompt) + output_observed = all(_ids(ids) for ids in output) + prompt_declared = [ + (result.get("usage") or {}).get("prompt_tokens") for result in (streamed, nonstreamed) + ] + prompt_complete = prompt_observed and all( + type(count) is int and count > 0 and len(ids) == count + for ids, count in zip(prompt, prompt_declared, strict=True) + ) + declared = [ + (result.get("usage") or {}).get("completion_tokens") for result in (streamed, nonstreamed) + ] + counts_observed = all(type(count) is int and count > 0 for count in declared) + output_complete = ( + output_observed + and counts_observed + and all(len(ids) == count for ids, count in zip(output, declared, strict=True)) + ) + prompt_equal = prompt_complete and prompt[0] == prompt[1] + output_equal = output_complete and output[0] == output[1] + content_equal = streamed["content"] == nonstreamed["content"] + reasoning_equal = streamed["reasoning"] == nonstreamed["reasoning"] + finish_equal = streamed["finish_reason"] == nonstreamed["finish_reason"] + + def tool_values(result): + return [ + {key: tool[key] for key in ("name", "arguments", "parsed_arguments")} + for tool in result["tools"] + ] + + tools_equal = tool_values(streamed) == tool_values(nonstreamed) + if not prompt_complete or not output_complete: + classification = "MISSING_RAW_TOKEN_OBSERVATION" + elif not prompt_equal: + classification = "PROMPT_TOKEN_DIFFERENCE" + elif not output_equal: + classification = "GENERATED_TOKEN_DIFFERENCE" + elif not (content_equal and reasoning_equal and finish_equal and tools_equal): + classification = "PARSED_OUTPUT_DIFFERENCE" + else: + classification = "EXACT_OBSERVED_MATCH" + + report = seal( + { + "schema": "urn:qwen:protocol-pair-comparison:v1", + "comparison_kind": kind, + "classification": classification, + "proof": "UNPROVED", + "scope": "This observed request pair; no universal parser or backend guarantee.", + "result_sha256": [digest(result) for result in (streamed, nonstreamed)], + "prompt_observed": prompt_observed, + "prompt_complete": prompt_complete, + "prompt_equal": prompt_equal, + "prompt_counts": [len(ids) if isinstance(ids, list) else None for ids in prompt], + "output_complete": output_complete, + "output_equal": output_equal, + "output_counts": [len(ids) if isinstance(ids, list) else None for ids in output], + "declared_completion_counts": declared, + "declared_prompt_counts": prompt_declared, + "first_prompt_difference": _first_difference(*prompt) if prompt_observed else None, + "first_output_difference": _first_difference(*output) if output_complete else None, + "content_equal": content_equal, + "reasoning_equal": reasoning_equal, + "finish_equal": finish_equal, + "tools_equal": tools_equal, + "content_lengths": [len(result["content"]) for result in (streamed, nonstreamed)], + "reasoning_lengths": [len(result["reasoning"]) for result in (streamed, nonstreamed)], + "content_equal_after_strip": streamed["content"].strip() + == nonstreamed["content"].strip(), + "whitespace_accepted_as_equal": False, + } + ) + write_private(evidence, report) + if classification != "EXACT_OBSERVED_MATCH": + raise DiagnosticError(f"response comparison ({kind}): {classification}") + return report diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_queue.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_queue.py new file mode 100644 index 0000000..1ab26d5 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_queue.py @@ -0,0 +1,164 @@ +"""Durable priority queue for the finite GPU campaign; no GPU imports.""" + +from __future__ import annotations + +import fcntl +import json +import os +import shutil +import time +from pathlib import Path + +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, private_json, seal, write_private + +STAGES = ("pilot", "focused", "extended") + + +def replace_private(root, name, document): + temporary = root / f".{name}.{os.getpid()}.{time.time_ns()}" + write_private(temporary, document) + temporary.replace(root / name) + descriptor = os.open(root, os.O_DIRECTORY) + try: + os.fsync(descriptor) + finally: + os.close(descriptor) + + +def space_required(case, reserve): + # Conservative state-export allowance, not a prediction of disk traffic. + # Long forced M1/D7 comparisons keep multiple complete state frames. + frames = 20 if case["family"] == "forced_d7" else 8 + capture = case["context"] * 32768 * frames + return reserve + max(4 * 1024**3, capture) + + +def request_pause(root): + root = Path(root).resolve() + private_json(root / "campaign.json") + replace_private(root, "pause-request.json", {"requested_ns": time.time_ns()}) + return {"pause_requested": True, "boundary": "after_current_case"} + + +def run( + campaign, + root, + *, + selected, + keep_going, + resume, + through, + budget_seconds, + retry_failed, + min_free_bytes, +): + from qwen_r9700_lab import conformance_campaign as runner + + if through not in STAGES or (budget_seconds is not None and budget_seconds <= 0): + raise DiagnosticError("invalid campaign stage or time budget") + if min_free_bytes < 0: + raise DiagnosticError("invalid free-space reserve") + root = Path(root).resolve() + if resume: + if not root.is_dir(): + raise DiagnosticError("cannot resume a missing campaign") + else: + root.mkdir(mode=0o700) + lock = os.open(root / "queue.lock", os.O_CREAT | os.O_RDWR | os.O_NOFOLLOW, 0o600) + try: + try: + fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError as error: + raise DiagnosticError("campaign already has an active controller") from error + if resume: + results = runner.load_results(campaign, root) + # An interrupted attempt is evidence, never silently a fresh case. + for attempt in sorted(root.glob("case-*")): + if not (attempt / "input.json").is_file() or (attempt / "result.json").exists(): + continue + payload = private_json(attempt / "input.json") + if payload["campaign"] != campaign: + raise DiagnosticError("interrupted attempt belongs to another campaign") + case = payload["case"] + result = runner.case_result( + campaign, + case, + "ERROR", + started=time.monotonic(), + error_type="InterruptedAttempt", + detail="Controller stopped before a durable result; artifacts retained.", + ) + write_private(attempt / "result.json", result) + results.setdefault(case["id"], []).append(result) + (root / "pause-request.json").unlink(missing_ok=True) + else: + write_private(root / "campaign.json", campaign) + results = {} + runner.coverage(campaign, results) # validate all resumed attempts first + started = time.monotonic() + + def checkpoint(status, current=None, **details): + report = runner.coverage(campaign, results) + replace_private(root, "coverage.json", report) + value = seal( + { + "schema": "urn:qwen:conformance-checkpoint:v1", + "campaign": campaign["sha256"], + "updated_ns": time.time_ns(), + "status": status, + "current": current, + "through": through, + "counts": report["counts"], + "elapsed_this_run": time.monotonic() - started, + **details, + } + ) + replace_private(root, "checkpoint.json", value) + print( + json.dumps({"status": status, "case": current, "counts": report["counts"]}), + flush=True, + ) + return report + + checkpoint("running") + for index, case in enumerate(campaign["cases"]): + if case["id"] not in selected or STAGES.index(case["stage"]) > STAGES.index(through): + continue + attempts = results.get(case["id"], []) + if attempts and attempts[-1]["status"] == "TESTED": + continue + if attempts and not retry_failed: + if not keep_going: + return checkpoint("failed_attempt_requires_review", case["id"]) + continue + if (root / "pause-request.json").exists(): + return checkpoint("paused", case["id"]) + from qwen_r9700_lab.conformance_gpu_lease import cleanup_block + + if cleanup_block() is not None: + return checkpoint("gpu_cleanup_incomplete", case["id"]) + if budget_seconds is not None and time.monotonic() - started >= budget_seconds: + return checkpoint("time_budget_reached", case["id"]) + required = space_required(case, min_free_bytes) + free = shutil.disk_usage(root).free + if free < required: + return checkpoint( + "insufficient_space", case["id"], free_bytes=free, required_bytes=required + ) + name = f"case-{index:05d}" + (f"-attempt-{len(attempts):03d}" if attempts else "") + case_root = root / name + case_root.mkdir(mode=0o700) + write_private(case_root / "input.json", {"campaign": campaign, "case": case}) + checkpoint("running_case", case["id"], attempt=name) + result = runner.execute_case(campaign, case, case_root) + write_private(case_root / "result.json", result) + results.setdefault(case["id"], []).append(result) + report = checkpoint("case_finished", case["id"]) + write_private(root / f"checkpoint-{name}.json", report) + if result["status"] != "TESTED" and not keep_going: + return checkpoint("failed", case["id"]) + return checkpoint( + "complete" if runner.coverage(campaign, results)["complete"] else "stage_finished" + ) + finally: + os.close(lock) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_radiance.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_radiance.py new file mode 100644 index 0000000..fcaef23 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_radiance.py @@ -0,0 +1,779 @@ +"""Explicitly armed, source-bound V2 Radiance adapter; imports no GPU code at rest. + +This is an isolated diagnostic worker adapter, never a production extension. +Its captures synchronize GPU state and are deliberately unsuitable for timing +qualification. Native execution remains UNPROVED until its GPU fault campaign. +""" + +from __future__ import annotations + +import contextlib +import hashlib +import inspect +import os +import re +import traceback +from pathlib import Path + +import numpy as np + +from qwen_r9700_lab.conformance_artifacts import capture_runtime +from qwen_r9700_lab.conformance_boundaries import BoundaryRecorder +from qwen_r9700_lab.conformance_dispatch import DispatchRecorder +from qwen_r9700_lab.conformance_model import state_names +from qwen_r9700_lab.conformance_reference import bf16, kv_scaling, reference_precision +from qwen_r9700_lab.conformance_replay import ( + OUTPUT_COMPONENTS, + CampaignWriter, + observation_domain, + validate_plan, +) +from qwen_r9700_lab.conformance_state import FrameWriter +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + digest, + integer, + private_json, + seal, + write_private, +) + + +def temporal_column(running_column, accepted_count): + return running_column + accepted_count - 1 + + +def convolution_offset(accepted_count): + return accepted_count - 1 + + +def gather_paged(cache, block_table, consumed, *, block_size, kv_heads, head_dim): + """Normalize R4D HND physical pages to logical [token, head, channel].""" + integer(consumed) + if cache.ndim != 4 or tuple(cache.shape[1:]) != (kv_heads, block_size, 2 * head_dim): + raise DiagnosticError("unsupported R4D cache layout") + table = np.asarray(block_table) + count = (consumed + block_size - 1) // block_size + if table.ndim != 1 or table.dtype.kind not in "iu" or len(table) < count: + raise DiagnosticError("incomplete logical page table") + ids = table[:count] + if np.any(ids < 0) or np.any(ids >= cache.shape[0]) or len(set(ids.tolist())) != count: + raise DiagnosticError("invalid or aliased pages within a writable sequence") + ordered = ( + np.asarray(cache[ids]).transpose(0, 2, 1, 3).reshape(-1, kv_heads, 2 * head_dim)[:consumed] + ) + return ordered[..., :head_dim].copy(), ordered[..., head_dim:].copy() + + +def gather_hybrid( + conv, temporal, table, *, running_column, accepted_count, history_width, dim_first +): + integer(running_column) + if not 1 <= integer(accepted_count) <= 8: + raise DiagnosticError("unsupported accepted-state width") + column = temporal_column(running_column, accepted_count) + if column >= len(table): + raise DiagnosticError("missing accepted recurrent-state version") + conv_id, temporal_id = int(table[running_column]), int(table[column]) + if not 0 <= conv_id < conv.shape[0] or not 0 <= temporal_id < temporal.shape[0]: + raise DiagnosticError("recurrent state references a nonexistent block") + window = np.asarray(conv[conv_id]) + if not dim_first: + window = window.T + start = convolution_offset(accepted_count) + if window.ndim != 2 or window.shape[1] < start + history_width: + raise DiagnosticError("convolution state misses the committed history window") + return np.asarray(temporal[temporal_id]).copy(), window[:, start : start + history_width].copy() + + +def verify_sources(package_root: Path, binding: dict): + authenticate(binding) + if binding.get("schema") != "urn:qwen:radiance-native-binding:v1" or not binding.get("files"): + raise DiagnosticError("missing pinned native source binding") + for name, expected in binding["files"].items(): + if Path(name).is_absolute() or ".." in Path(name).parts: + raise DiagnosticError("unsafe native binding path") + path = package_root / name + if hashlib.sha256(path.read_bytes()).hexdigest() != expected: + raise DiagnosticError("native source changed; adapter binding needs review") + return binding["sha256"] + + +_KV_FORMATS = { + "bf16": ({"auto", "bfloat16"}, {"torch.bfloat16"}), + "fp8_e4m3fn": ({"fp8", "fp8_e4m3"}, {"torch.uint8", "torch.float8_e4m3fn"}), +} + + +def validate_native_kv_encoding(cache_dtype, storage_dtype, reference_encoding): + # The source-bound R4D backend interprets fp8/fp8_e4m3 as E4M3FN even + # though vLLM allocates its backing tensors as uint8. Other byte-backed + # formats (E5M2, per-token scaling, etc.) are not interchangeable. + formats, storage_types = _KV_FORMATS.get(reference_encoding, (set(), set())) + if cache_dtype not in formats or str(storage_dtype) not in storage_types: + raise DiagnosticError( + f"native KV encoding differs from the declared reference: " + f"cache={cache_dtype}, storage={storage_dtype}, reference={reference_encoding}" + ) + + +def as_cpu(tensor, *, storage=False, kv_encoding=None): + # Importing this module on the host never imports torch or queries a device. + import torch + + if storage: + declared = kv_encoding or { + "torch.float8_e4m3fn": "fp8_e4m3fn", + "torch.bfloat16": "bf16", + }.get(str(tensor.dtype)) + storage_types = _KV_FORMATS.get(declared, (set(), set()))[1] + if str(tensor.dtype) not in storage_types: + raise DiagnosticError("native KV storage must be declared BF16 or E4M3FN") + encoding = torch.uint8 if declared == "fp8_e4m3fn" else torch.uint16 + return tensor.detach().view(encoding).contiguous().cpu().numpy().copy() + return tensor.detach().float().contiguous().cpu().numpy().copy() + + +def export_call_tensor(tensor): + """Raw storage for semantic call evidence; preserve integer and FP8 bits.""" + import torch + + value = tensor.detach().contiguous() + if value.dtype == torch.bfloat16: + value = value.view(torch.uint16) + elif str(value.dtype).startswith("torch.float8"): + value = value.view(torch.uint8) + return value.cpu().numpy().copy() + + +class ConformanceWorkerExtension: + """Named worker RPCs using vLLM's standard serialization contract.""" + + def qwen_conformance_install(self, plan_path: str, output_path: str, binding_path: str): + return install(self, plan_path, output_path, binding_path) + + def qwen_conformance_finish(self): + return finish(self) + + +def preserve_native_failure(root, plan, binding, stage, error): + """Preserve the worker-side cause before the transport reduces it to engine death.""" + document = seal( + { + "schema": "urn:qwen:native-worker-failure:v1", + "plan": plan["sha256"], + "binding": binding["sha256"], + "stage": stage, + "type": type(error).__name__, + "message": str(error), + "traceback": "".join(traceback.format_exception(error)), + } + ) + # Preserve the first cause, including during failed teardown. + with contextlib.suppress(FileExistsError): + write_private(root / "worker-error.json", document) + + +def diagnosed_native_call(function, root, plan, binding, stage): + from functools import wraps + + @wraps(function) + def call(*args, **kwargs): + try: + return function(*args, **kwargs) + except Exception as error: + preserve_native_failure(root, plan, binding, stage, error) + raise + + return call + + +def install(worker, plan_path: str, output_path: str, binding_path: str): + """Entry point for LLM.collective_rpc, only in the isolated GPU worker.""" + if os.environ.get("QWEN_CONFORMANCE_GPU") != "1": + raise DiagnosticError("GPU instrumentation is not armed") + runner = worker.model_runner + package = Path(inspect.getfile(type(runner))).resolve().parents[5] + # .../site-packages/vllm/v1/worker/gpu/model_runner.py + if package.name != "site-packages": + package = next( + ( + p + for p in Path(inspect.getfile(type(runner))).resolve().parents + if p.name == "site-packages" + ), + None, + ) + if package is None: + raise DiagnosticError("cannot locate the pinned runtime source") + binding = private_json(Path(binding_path)) + verify_sources(package, binding) + if getattr(runner, "_qwen_conformance_probe", None) is not None: + raise DiagnosticError("native conformance probe is already installed") + probe = RadianceProbe( + runner, validate_plan(private_json(Path(plan_path))), Path(output_path), binding + ) + try: + probe.attach() + # Capture at the real worker entry points, outside all nested observation + # hooks. A transport-level EngineDeadError is insufficient for a negative control. + for method in ("execute_model", "sample_tokens"): + probe.hooks.replace( + runner, + method, + diagnosed_native_call( + getattr(runner, method), probe.campaign.root, probe.plan, binding, method + ), + ) + except BaseException as error: + try: + preserve_native_failure(probe.campaign.root, probe.plan, binding, "install", error) + finally: + probe.detach() + raise + runner._qwen_conformance_probe = probe + return {"installed": True, "binding": binding["sha256"], "gpu_qualification": "UNPROVED"} + + +def finish(worker): + runner = worker.model_runner + try: + return runner._qwen_conformance_probe.finish() + finally: + del runner._qwen_conformance_probe + + +def capture_committed_state( + worker, *, output_path, request_id, expected, plan_path, binding_path, quiescent=False +): + """Explicit worker RPC for live/restored V2 state, independent of forced replay. + + The caller must hold scheduler admission and wait for any connector restore + or bank handover to finish before setting quiescent. A GPU synchronization + alone cannot establish that CPU-side ownership will stay unchanged. The + requested prefix/count/pending identity is checked before a frame is saved. + No generation or model output is forced and no snapshot is loaded/saved here. + """ + from types import SimpleNamespace + + from qwen_r9700_lab.conformance_state import read_frame + + if os.environ.get("QWEN_CONFORMANCE_GPU") != "1" or quiescent is not True: + raise DiagnosticError("native state capture requires an armed, quiescent worker boundary") + runner = worker.model_runner + package = next( + ( + p + for p in Path(inspect.getfile(type(runner))).resolve().parents + if p.name == "site-packages" + ), + None, + ) + if package is None: + raise DiagnosticError("cannot locate pinned native state producer") + binding = private_json(Path(binding_path)) + verify_sources(package, binding) + plan = validate_plan(private_json(Path(plan_path))) + precision = reference_precision(plan.get("reference_profile", "radiance-fp8")) + if Path(runner.model_config.model).resolve() != Path(plan["checkpoint"]).resolve(): + raise DiagnosticError("native state producer loaded a different checkpoint path") + config = runner.model_config.hf_text_config.to_dict() + kv_scaling(config, plan["kv_scales"], precision["profile"]) + validate_native_kv_encoding( + runner.cache_config.cache_dtype, runner.kv_cache_dtype, precision["kv_encoding"] + ) + parallel = runner.vllm_config.parallel_config + if ( + config["model_type"] != "qwen3_5_text" + or runner.cache_config.mamba_cache_mode != "align" + or parallel.tensor_parallel_size != 1 + or parallel.data_parallel_size != 1 + or parallel.pipeline_parallel_size != 1 + ): + raise DiagnosticError("unadmitted native restore layout") + if set(expected) != {"consumed", "pending", "input_digest"}: + raise DiagnosticError("recovery capture requires exact position, pending token and prefix") + index = runner.req_states.req_id_to_index.get(request_id) + if index is None: + raise DiagnosticError("requested native sequence is not resident") + output = Path(output_path) + # Reuse the same extraction code qualified by the independent replay, while + # keeping its source and evidence identities distinct from forced execution. + probe = RadianceProbe.__new__(RadianceProbe) + probe.runner, probe.config, probe.plan = runner, config, plan + probe.adapter = hashlib.sha256(Path(__file__).read_bytes()).hexdigest() + probe.execution = digest( + { + "plan": plan["execution"], + "source": binding["sha256"], + "capture": "quiescent-native-state", + } + ) + probe.campaign = SimpleNamespace( + root=output.parent, coverage=state_names(config), record=lambda _: None + ) + probe.capture( + SimpleNamespace(idx_mapping_np=[index]), + {**expected, "phase": "restore", "name": output.name}, + None, + mode="native_recovery_capture", + ) + if runner.req_states.req_id_to_index.get(request_id) != index: + raise DiagnosticError("sequence ownership changed during native state capture") + frame = read_frame(output) + return { + "frame": frame["sha256"], + "consumed": frame["consumed"], + "native_extraction": "UNPROVED", + "quiescence": "ASSUMED: caller-held admission", + } + + +class RadianceProbe: + def __init__(self, runner, plan, root, binding): + self.runner, self.plan, self.binding = runner, plan, binding + precision = reference_precision(plan.get("reference_profile", "radiance-fp8")) + c = runner.model_config.hf_text_config + self.config = c.to_dict() + kv_scaling(self.config, plan["kv_scales"], precision["profile"]) + parallel = runner.vllm_config.parallel_config + if ( + c.model_type != "qwen3_5_text" + or parallel.tensor_parallel_size != 1 + or parallel.data_parallel_size != 1 + or parallel.pipeline_parallel_size != 1 + ): + raise DiagnosticError("native adapter admits one dense Qwen3.5 TP1/DP1 worker") + if ( + not runner.model_config.enforce_eager + or runner.vllm_config.scheduler_config.async_scheduling + ): + raise DiagnosticError("full tensor capture requires explicitly serialized eager replay") + if runner.cache_config.mamba_cache_mode != "align": + raise DiagnosticError( + "native hybrid adapter requires the reviewed align state convention" + ) + if runner.vllm_config.kv_transfer_config is not None: + raise DiagnosticError("replay must start independently with no snapshot connector") + validate_native_kv_encoding( + runner.cache_config.cache_dtype, runner.kv_cache_dtype, precision["kv_encoding"] + ) + self.campaign = CampaignWriter( + root, + plan, + coverage=state_names(self.config) + OUTPUT_COMPONENTS, + backend="radiance-v2-native-candidate", + ) + self.adapter = hashlib.sha256(Path(__file__).read_bytes()).hexdigest() + self.artifacts = capture_runtime() + write_private(root / "runtime-before.json", self.artifacts) + self.execution = digest( + { + "plan_execution": plan["execution"], + "binding": binding["sha256"], + "adapter": self.adapter, + "schedule": "serialized-eager", + "runtime_artifacts": self.artifacts["sha256"], + } + ) + positions, inputs = observation_domain(plan) + self.boundaries = BoundaryRecorder( + root / "boundaries", + contract=plan["contract"], + execution=self.execution, + adapter=self.adapter, + positions=positions, + layers=c.num_hidden_layers, + input_digests=inputs, + ) + self.pending_capture, self.logits, self.batch = None, None, None + self.index, self.cursor, self.handles = 0, 0, [] + self.positions = None + from qwen_r9700_lab.conformance_instrumentation import CallRecorder, HookSet + + self.hooks = HookSet() + self.calls = CallRecorder( + root / "calls", + contract=plan["contract"], + execution=self.execution, + adapter=self.adapter, + tensor_export=export_call_tensor, + is_tensor=lambda value: hasattr(value, "detach") and hasattr(value, "dtype"), + mode=os.environ.get("QWEN_CONFORMANCE_CALL_MODE", "tensor"), + ) + self.dispatch = None + + def detach(self): + try: + for handle in reversed(self.handles): + handle.remove() + finally: + self.handles.clear() + try: + if getattr(self, "dispatch", None) is not None: + self.dispatch.close() + finally: + self.hooks.close() + + def attach(self): + from qwen_r9700_lab.conformance_faults import install_experiment + + install_experiment(self) + runner = self.runner + original_sample, original_post = runner.sample, runner.postprocess_sampled + original_logits = runner.model.compute_logits + original_prepare = runner.prepare_inputs + + def prepare(*args, **kwargs): + batch = original_prepare(*args, **kwargs) + if batch.num_reqs != 1: + raise DiagnosticError("native replay observed an unadmitted concurrent request") + self.positions = batch.positions[: batch.num_tokens].detach().cpu().numpy().tolist() + return batch + + def logits(*args, **kwargs): + out = original_logits(*args, **kwargs) + self.logits = as_cpu(out) + return out + + def sample(hidden_states, batch, grammar_output): + if grammar_output is not None: + raise DiagnosticError( + "grammar and parser semantics need a separate checked protocol adapter" + ) + result, ns, nr = original_sample(hidden_states, batch, grammar_output) + n = int(ns[0].item()) + if not n: # intermediate prefill chunk: no pending output yet + return result, ns, nr + if self.index >= len(self.campaign.expected): + raise DiagnosticError("native backend exceeded its replay schedule") + expected = self.campaign.expected[self.index] + drafts = int(batch.num_draft_tokens) + widths = self.plan.get("accepted_widths", [0] * (len(self.plan["forced_tokens"]) - 1)) + accepted = widths[self.index - 1] if self.index else 0 + if accepted > drafts or (drafts and drafts != 7): + raise DiagnosticError("native verifier width is outside the admitted D7 schedule") + # Capture genuine target results before injecting the diagnostic + # accept boundary. The forced decision is never sold as a sample. + if self.logits is None or self.logits.shape[0] != drafts + 1: + raise DiagnosticError("missing causal target verification rows") + natural = self.logits[accepted].copy() + begin = self.cursor + (1 if self.index else 0) + count = accepted + 1 if self.index else 1 + forced = self.plan["forced_tokens"][begin : begin + count] + if len(forced) != count: + raise DiagnosticError("missing forced output suffix") + result.sampled_token_ids.fill_(-1) + for j, token in enumerate(forced): + result.sampled_token_ids[0, j] = token + ns.fill_(count) + nr.fill_(drafts - accepted) + self.cursor = expected["consumed"] - len(self.plan["prefix"]) + self.next_proposal_accept = widths[self.index] if self.index < len(widths) else 0 + self.pending_capture = (batch, expected, natural) + return result, ns, nr + + def post(*args, **kwargs): + result = original_post(*args, **kwargs) + if self.pending_capture is not None: + batch, expected, target_logits = self.pending_capture + self.capture(batch, expected, target_logits) + self.pending_capture = None + self.index += 1 + return result + + self.hooks.replace(runner, "prepare_inputs", prepare) + self.hooks.replace(runner, "sample", sample) + self.hooks.replace(runner, "postprocess_sampled", post) + self.hooks.replace(runner.model, "compute_logits", logits) + if runner.speculator is not None: + original_propose = runner.speculator.propose + + def propose(*args, **kwargs): + out = original_propose(*args, **kwargs) + if out.ndim != 2 or out.shape[0] != 1 or out.shape[1] != 7: + raise DiagnosticError("unexpected native drafter representation") + # Proposals after the pending token. A rejected suffix beyond + # the plan uses public token 0, never another session's data. + for j in range(7): + p = self.cursor + 1 + j + out[0, j] = ( + self.plan["forced_tokens"][p] if p < len(self.plan["forced_tokens"]) else 0 + ) + # Perturb only proposals that the NEXT forced commit will + # reject. Accepted prefix and pending-token inputs stay fixed. + next_width = getattr(self, "next_proposal_accept", 0) + if j >= next_width and hasattr(self, "rejected_suffix_token"): + out[0, j] = self.rejected_suffix_token + return out + + self.hooks.replace(runner.speculator, "propose", propose) + + layers = [ + (int(m.group(1)), module) + for name, module in runner.model.named_modules() + if (m := re.search(r"(?:^|\.)layers\.(\d+)$", name)) + ] + if len(layers) != self.config["num_hidden_layers"]: + raise DiagnosticError("native decoder inventory is incomplete") + for layer, module in layers: + for name, stage in ( + ("input_layernorm", "input_norm"), + ("post_attention_layernorm", "post_attention_norm"), + ): + child = getattr(module, name) + + def observed(mod, args, output, layer=layer, stage=stage): + value = output[0] if isinstance(output, tuple) else output + self.record_boundary(layer, stage, as_cpu(value)) + + self.handles.append(child.register_forward_hook(observed)) + + def layer_output(mod, args, output, layer=layer): + if not isinstance(output, tuple) or len(output) != 2: + raise DiagnosticError("native residual boundary changed") + value = bf16(as_cpu(output[0]) + as_cpu(output[1])) + self.record_boundary(layer, "output", value) + + self.handles.append(module.register_forward_hook(layer_output)) + + def call_context(): + if self.positions is None or self.index >= len(self.campaign.expected): + return None + expected = self.campaign.expected[self.index] + if not any( + p < expected["consumed"] and p in self.boundaries.position_set + for p in self.positions + ): + return None + return { + "consumed": expected["consumed"], + "input_digest": expected["input_digest"], + "positions": self.positions, + "phase": expected["phase"], + } + + # Observe actual target module calls and numerical glue. Aliases hidden + # inside compiled operators remain unqualified; calls.json lists what + # really executed. No import-time or production hook installation. + for name, module in runner.model.named_modules(): + site = "target." + (name or "root") + self.calls.bind(module, "forward", site=site, hooks=self.hooks, context=call_context) + if re.search(r"(?:^|\.)layers\.\d+$", name): + self.calls.required.add(site) + import sys + + # An explicit reviewed export inventory is required. Absence is recorded + # as a gap; never infer complete device coverage from module hooks. + entries = self.binding.get("native_entrypoints", []) + if entries: + self.dispatch = DispatchRecorder( + self.campaign.root / "dispatch", execution=self.execution + ) + for entry in entries: + name = entry["binding"]["module"] + module = sys.modules.get(name) + aliases = [sys.modules.get(alias) for alias in entry["aliases"]] + if module is None or any(alias is None for alias in aliases): + raise DiagnosticError("native dispatch binding module or alias is not loaded") + self.dispatch.bind(module, entry["binding"], aliases=aliases) + + for module_name, functions in { + "radiance_gdn": ( + "conv_prep", + "conv_update", + "fused_update", + "kkt_solve", + "output_norm", + "recurrent_update", + "fused_prefill", + ), + "radiance_mxfp4": ("mxfp4_linear", "mxfp4_linear_pq"), + }.items(): + module = sys.modules.get(module_name) + if module is None: + continue + for name in functions: + self.calls.bind( + module, + name, + site=module_name + "." + name, + hooks=self.hooks, + context=call_context, + ) + + def record_boundary(self, layer, stage, values): + if self.positions is None or values.shape[0] != len(self.positions): + raise DiagnosticError("native boundary has no exact token positions") + for row, position in enumerate(self.positions): + if ( + self.index < len(self.campaign.expected) + and position < self.campaign.expected[self.index]["consumed"] + ): + self.boundaries.record(position, layer, stage, values[row]) + + def capture(self, batch, expected, target_logits, *, mode="forced_token_replay"): + import torch + from vllm.model_executor.layers.mamba.mamba_utils import is_conv_state_dim_first + + torch.cuda.synchronize() # diagnostics only, not a production hook + runner, c = self.runner, self.config + precision = reference_precision(self.plan.get("reference_profile", "radiance-fp8")) + index = int(batch.idx_mapping_np[0]) + consumed = int(runner.req_states.num_computed_tokens.gpu[index].item()) + pending = int(runner.req_states.last_sampled_tokens[index, 0].item()) + tokens = runner.req_states.all_token_ids.gpu[index, :consumed].cpu().numpy().astype("= len(table): + raise DiagnosticError("invalid native committed state version") + # Copy only the selected bank/window, not the whole GPU pool. + selected_conv = as_cpu(conv[int(table[running])]) + if not is_conv_state_dim_first(): + selected_conv = selected_conv.T + offset = convolution_offset(accepted) + history = selected_conv[:, offset : offset + c["linear_conv_kernel_dim"] - 1] + if history.shape[-1] != c["linear_conv_kernel_dim"] - 1: + raise DiagnosticError("native convolution window is incomplete") + writer.array(prefix + "gdn", as_cpu(state[int(table[column])])) + writer.array(prefix + "conv", history) + else: + cache = module.kv_cache + heads, width = c["num_key_value_heads"], c["head_dim"] + if cache.ndim != 4 or cache.shape[1] != heads or cache.shape[3] != width * 2: + raise DiagnosticError("native R4D HND cache layout changed") + block_size = cache.shape[2] + needed = (consumed + block_size - 1) // block_size + ids = table[:needed] + if ( + len(ids) != needed + or np.any(ids < 0) + or np.any(ids >= cache.shape[0]) + or len(set(ids.tolist())) != needed + ): + raise DiagnosticError("invalid native page ownership") + parts = [ + as_cpu( + cache[int(b)], storage=True, kv_encoding=precision["kv_encoding"] + ).transpose(1, 0, 2) + for b in ids + ] + logical = np.concatenate(parts, axis=0)[:consumed] + for name, data in ( + ("keys", logical[..., :width]), + ("values", logical[..., width:]), + ): + writer.add( + prefix + name, + np.ascontiguousarray(data).tobytes(), + dtype=precision["kv_encoding"], + shape=data.shape, + ) + scales = np.asarray( + [module._k_scale_float, module._v_scale_float] + if precision["kv_fp8"] + else [1.0, 1.0], + dtype="> 4) & 15 + + +def canonical_slot(block, offset, block_size): + return block * block_size + offset + + +def bf16_round_bits(bits): + """RNE bit expression for non-NaN FP32; also executed symbolically by Z3.""" + return (bits + 0x7FFF + ((bits >> 16) & 1)) & 0xFFFF0000 + + +def bf16(value): + x = np.asarray(value, dtype=np.float32) + bits = x.view(np.uint32) + rounded = bf16_round_bits(bits) + # Preserve NaN as NaN, including small NaN payloads that would round to Inf. + nan = ((bits & 0x7F800000) == 0x7F800000) & ((bits & 0x007FFFFF) != 0) + rounded = np.where(nan, bits | 0x00400000, rounded).astype(np.uint32) + return (rounded & np.uint32(0xFFFF0000)).view(np.float32) + + +def fp8_encode(value): + """Saturating finite E4M3FN, round-to-nearest, ties to an even code.""" + x = np.asarray(value, dtype=np.float32) + if not np.isfinite(x).all(): + raise DiagnosticError("reference FP8 quantizer rejects nonfinite inputs") + magnitude = np.minimum(np.abs(x), np.float32(448)) + hi = np.minimum(np.searchsorted(FP8_POSITIVE, magnitude), 126) + lo = np.maximum(hi - 1, 0) + lower, upper = magnitude - FP8_POSITIVE[lo], FP8_POSITIVE[hi] - magnitude + choose_hi = (upper < lower) | ((upper == lower) & ((hi & 1) == 0)) + code = np.where(choose_hi, hi, lo).astype(np.uint8) + return code | (np.signbit(x).astype(np.uint8) << 7) + + +def fp8_decode(code): + a = np.asarray(code, dtype=np.uint8) + return values(a.tobytes(), "fp8_e4m3fn").astype(np.float32).reshape(a.shape) + + +def activation_quantize(value): + x = np.asarray(value, dtype=np.float32) + if not x.size or not np.isfinite(x).all(): + raise DiagnosticError("invalid activation quantizer input") + scale = np.maximum( + np.max(np.abs(x), axis=-1, keepdims=True) / np.float32(448), np.float32(1 / (448 * 512)) + ) + code = fp8_encode(np.divide(x, scale, dtype=np.float32)) + return code, scale + + +def unpack_mxfp4(packed, scales): + packed, scales = np.asarray(packed), np.asarray(scales) + if packed.dtype != np.uint8 or scales.dtype != np.uint8 or packed.ndim != 2: + raise DiagnosticError("MXFP4 storage must be a packed matrix with E8M0 scales") + n, half_k = packed.shape + if half_k * 2 % 32 or scales.shape != (n, half_k * 2 // 32) or np.any(scales == 255): + raise DiagnosticError("unsupported MXFP4 shape or nonfinite E8M0 scale") + codes = np.stack((low_nibble(packed), high_nibble(packed)), axis=-1).reshape(n, -1) + expanded = np.repeat( + np.ldexp(np.ones_like(scales, dtype=np.float32), scales.astype(np.int32) - 127), 32, axis=-1 + ) + return np.multiply(LEVELS[codes], expanded, dtype=np.float32) + + +def ordered_sum(value, axis=-1): + x = np.moveaxis(np.asarray(value, dtype=np.float32), axis, -1) + result = np.zeros(x.shape[:-1], dtype=np.float32) + for i in range(x.shape[-1]): + result = np.add(result, x[..., i], dtype=np.float32) + return result + + +def linear(value, weight, *, quantize_activation=False, output_bf16=True): + x, w = np.asarray(value, dtype=np.float32), np.asarray(weight, dtype=np.float32) + if x.shape[-1] != w.shape[-1] or w.ndim != 2: + raise DiagnosticError("linear dimensions do not agree") + if quantize_activation: + code, scale = activation_quantize(x) + x = np.multiply(fp8_decode(code), scale, dtype=np.float32) + result = np.zeros((*x.shape[:-1], w.shape[0]), dtype=np.float32) + for k in range(w.shape[-1]): + product = np.multiply(x[..., k, None], w[:, k], dtype=np.float32) + result = np.add(result, product, dtype=np.float32) + return bf16(result) if output_bf16 else result + + +def rms_norm(value, weight, epsilon, *, weight_offset=0.0, output_bf16=True): + x = np.asarray(value, dtype=np.float32) + variance = ordered_sum(np.multiply(x, x, dtype=np.float32)) / np.float32(x.shape[-1]) + inverse = np.reciprocal(np.sqrt(variance + np.float32(epsilon), dtype=np.float32)) + normalized = np.multiply(x, inverse[..., None], dtype=np.float32) + output = np.multiply( + normalized, + np.asarray(weight, dtype=np.float32) + np.float32(weight_offset), + dtype=np.float32, + ) + return bf16(output) if output_bf16 else output + + +def sigmoid(x): + x = np.asarray(x, dtype=np.float32) + e = np.exp(-np.abs(x), dtype=np.float32) + return np.where(x >= 0, 1 / (1 + e), e / (1 + e)).astype(np.float32) + + +def silu(x): + return np.multiply(np.asarray(x, dtype=np.float32), sigmoid(x), dtype=np.float32) + + +def softplus(x): + x = np.asarray(x, dtype=np.float32) + return np.add( + np.maximum(x, np.float32(0)), + np.log1p(np.exp(-np.abs(x), dtype=np.float32)), + dtype=np.float32, + ) + + +def rope(value, position, rotary_dim, theta): + x = np.asarray(value, dtype=np.float32).copy() + if rotary_dim % 2 or rotary_dim > x.shape[-1] or position < 0: + raise DiagnosticError("invalid text rotary position geometry") + frequencies = np.float32(position) / np.power( + np.float32(theta), np.arange(0, rotary_dim, 2, dtype=np.float32) / np.float32(rotary_dim) + ) + cosine, sine = bf16(np.cos(frequencies)), bf16(np.sin(frequencies)) + half = rotary_dim // 2 + a, b = x[..., :half].copy(), x[..., half:rotary_dim].copy() + x[..., :half] = np.subtract(np.multiply(a, cosine), np.multiply(b, sine), dtype=np.float32) + x[..., half:rotary_dim] = np.add(np.multiply(b, cosine), np.multiply(a, sine), dtype=np.float32) + return bf16(x) + + +def convolution_step(value, history, weight): + x, old, w = (np.asarray(v, dtype=np.float32) for v in (value, history, weight)) + if w.ndim != 2 or old.shape != (w.shape[0], w.shape[1] - 1) or x.shape != (w.shape[0],): + raise DiagnosticError("invalid causal convolution geometry") + window = np.concatenate((old, x[:, None]), axis=1) + output = bf16(silu(ordered_sum(np.multiply(window, w, dtype=np.float32)))) + return output, window[:, 1:].copy() + + +def gdn_step(q, k, v, decay_log, beta, state, *, scale=None): + q, k, v, decay_log, beta, state = ( + np.asarray(x, dtype=np.float32) for x in (q, k, v, decay_log, beta, state) + ) + heads, value_width, key_width = state.shape + if ( + heads % q.shape[0] + or q.shape != k.shape + or q.shape[1] != key_width + or v.shape != (heads, value_width) + ): + raise DiagnosticError("invalid GDN geometry") + if decay_log.shape != (heads,) or beta.shape != (heads,): + raise DiagnosticError("invalid GDN gates") + q, k = (np.repeat(x, heads // x.shape[0], axis=0) for x in (q, k)) + decayed = np.multiply( + state, np.exp(decay_log, dtype=np.float32)[:, None, None], dtype=np.float32 + ) + prediction = ordered_sum(np.multiply(decayed, k[:, None, :], dtype=np.float32)) + residual = np.subtract(v, prediction, dtype=np.float32) + update = np.multiply( + np.multiply(beta[:, None], residual, dtype=np.float32)[:, :, None], + k[:, None, :], + dtype=np.float32, + ) + new_state = np.add(decayed, update, dtype=np.float32) + output = ordered_sum(np.multiply(new_state, q[:, None, :], dtype=np.float32)) + output = np.multiply( + output, np.float32(key_width**-0.5 if scale is None else scale), dtype=np.float32 + ) + return bf16(output), new_state + + +def dense_attention(q, keys, vals, *, scale=None): + q, keys, vals = (np.asarray(v, dtype=np.float32) for v in (q, keys, vals)) + if q.ndim != 2 or keys.ndim != 3 or keys.shape != vals.shape: + raise DiagnosticError("invalid attention geometry") + heads, width = q.shape + if width != keys.shape[-1] or heads % keys.shape[1] or keys.shape[0] == 0: + raise DiagnosticError("invalid grouped attention geometry") + mapping = np.arange(heads) // (heads // keys.shape[1]) + scores = ordered_sum(keys[:, mapping, :] * q[None, :, :]) + scores = np.multiply(scores, np.float32(width**-0.5 if scale is None else scale)) + probabilities = np.exp(scores - np.max(scores, axis=0), dtype=np.float32) + probabilities /= ordered_sum(probabilities, axis=0)[None, :] + output = ordered_sum(probabilities[:, :, None] * vals[:, mapping, :], axis=0) + return bf16(output) + + +OPERATORS = { + "rms_norm": rms_norm, + "linear": linear, + "rope": rope, + "convolution_step": convolution_step, + "gdn_step": gdn_step, + "dense_attention": dense_attention, + "activation_quantize": activation_quantize, + "unpack_mxfp4": unpack_mxfp4, +} + + +REFERENCE_PROFILES = { + "weight-only-bf16": (False, False), + "weight-fp8-activations": (True, False), + "weight-fp8-kv": (False, True), + "radiance-fp8": (True, True), +} + + +def reference_precision(profile): + if not isinstance(profile, str) or profile not in REFERENCE_PROFILES: + raise DiagnosticError("unsupported reference precision profile") + activation_fp8, kv_fp8 = REFERENCE_PROFILES[profile] + return { + "profile": profile, + "activation_fp8": activation_fp8, + "kv_fp8": kv_fp8, + "kv_encoding": "fp8_e4m3fn" if kv_fp8 else "bf16", + "additional_quantization": [ + name for name, enabled in (("activations", activation_fp8), ("KV", kv_fp8)) if enabled + ], + } + + +def kv_scaling(config, kv_scales, profile): + precision = reference_precision(profile) + expected = {str(i) for i, kind in enumerate(config["layer_types"]) if kind == "full_attention"} + if not isinstance(kv_scales, dict): + raise DiagnosticError("KV scales must be an explicit layer mapping") + if not precision["kv_fp8"]: + units = {i: [1.0, 1.0] for i in expected} + if kv_scales and kv_scales != units: + raise DiagnosticError("BF16 KV has no quantizer; scales must be absent or unit") + return units + if set(kv_scales) != expected or any( + not isinstance(row, list) + or len(row) != 2 + or any( + type(value) not in (int, float) or not np.isfinite(value) or value <= 0 for value in row + ) + for row in kv_scales.values() + ): + raise DiagnosticError("all KV scale values must be declared explicitly") + return kv_scales + + +def reference_contract(profile="radiance-fp8"): + return { + "implementation": "numpy-serial-fp32-v1", + "numpy": np.__version__, + "reductions": "left-to-right separate FP32 multiply/add; no reassociation or FMA", + "transcendentals": "pinned NumPy float32 ufuncs; libm and compiler trusted", + "bf16": "round-to-nearest ties-even at explicit boundaries", + "fp8": "E4M3FN saturating nearest-even; per-row max(abs(x))/448; scale floor 1/(448*512)", + "mxfp4": "low nibble first; E2M1; E8M0/32; exponent bias127; reject scale255", + "attention": "native dense causal GQA; logical positions only; no Quest approximation", + "sampler": "greedy, first vocabulary index wins an exact tie", + "reference_implementation": "TESTED", + "native_equivalence": "UNPROVED", + "precision": reference_precision(profile), + "stock_implementation_equivalence": "UNPROVED; canonical arithmetic is not stock dispatch", + } + + +def reference_semantics(files, config, kv_scales, profile): + """Bind model choices independently of a caller-supplied contract digest.""" + arithmetic = reference_contract(profile) + precision = arithmetic["precision"] + return { + "weights": {"files": files, "config": config}, + "weight_quantization": { + "format": "MXFP4 E2M1/E8M0", + "group": 32, + "nibble_order": "low_first", + }, + "activation_quantization": ( + {"format": "E4M3FN", "policy": arithmetic["fp8"]} + if precision["activation_fp8"] + else {"format": "BF16", "extra_quantizer": False} + ), + "attention": {"method": "native dense causal GQA", "sparse_approximation": False}, + "kv_representation": { + "format": precision["kv_encoding"], + "scales": kv_scaling(config, kv_scales, profile), + }, + "recurrence": { + "state": "FP32", + "conv_history": "BF16 values", + "algorithm": "serial gated delta", + }, + "position_encoding": {"text_rope": config.get("rope_parameters")}, + "tokenizer": {"domain": "explicit token IDs; tokenization itself outside this replay"}, + "chat_template": { + "domain": "supplied rendered prefix; template construction outside this replay" + }, + "sampler": { + "method": "greedy full vocabulary", + "tie_break": "lowest token ID", + "forced_replay": "diagnostic inputs only; never published as model decisions", + }, + "numerical_contract": arithmetic, + } diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_reference_store.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_reference_store.py new file mode 100644 index 0000000..d44d48d --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_reference_store.py @@ -0,0 +1,318 @@ +"""One-slot store for independently retained serial qualification baselines. + +This owns disposable copies only. Native execution/configuration admission is +the caller's responsibility and must be included in the sealed identity. Every +hit is projected and byte-authenticated again. An interrupted publication or +retirement is an explicit error requiring review, never an implicit cache hit. +""" + +import fcntl +import hashlib +import os +import re +import shutil +import stat +import uuid +from pathlib import Path + +from qwen_r9700_lab.conformance_native_reference import ( + compatible_serial_plans, + project_serial_reference, +) +from qwen_r9700_lab.conformance_state import compare_frames, private_directory +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + digest, + private_json, + require_sha, + seal, + write_private, +) + +SCHEMA = "urn:qwen:serial-reference-store:v1" + + +def require(condition, message): + if not condition: + raise DiagnosticError(message) + + +def sync(path): + fd = os.open(path, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + try: + os.fsync(fd) + finally: + os.close(fd) + + +def sealed_file(path): + value = private_json(path) + authenticate(value) + return value + + +def copied_tree_identity(root): + """Bind a case-local source/JIT tree by all pre-execution bytes and modes. + + Absolute relocation and copy timestamps are not semantic inputs. Symlinks, + special files, empty trees and concurrent changes are refused, not omitted. + This is an artifact identity, not a proof about future compiler output. + """ + from qwen_r9700_lab.conformance_artifacts import file_identity + + root = Path(root) + + def inventory(): + entries = {} + for path in [root, *sorted(root.rglob("*"))]: + info = path.lstat() + require( + stat.S_ISDIR(info.st_mode) or stat.S_ISREG(info.st_mode), + "copied runtime tree contains a symlink or special file", + ) + require(info.st_uid == os.getuid(), "copied runtime tree must be owned") + entries[str(path.relative_to(root))] = tuple( + getattr(info, field) + for field in ( + "st_dev", + "st_ino", + "st_mode", + "st_size", + "st_mtime_ns", + "st_ctime_ns", + ) + ) + require(stat.S_ISDIR(entries["."][2]), "copied runtime tree is not a directory") + return entries + + before = inventory() + records = {} + files = 0 + for relative, metadata in before.items(): + mode = metadata[2] + records[relative] = {"mode": mode} + if stat.S_ISREG(mode): + records[relative].update(file_identity(root / relative)) + files += 1 + require(files > 0, "copied runtime tree is empty") + require(inventory() == before, "copied runtime tree changed during hashing") + return digest({"schema": SCHEMA + "/copied-tree", "entries": records}) + + +class SerialReferenceStore: + """Use under one explicit lock; retire before populating another baseline. + + Keeping replacement population outside this store avoids an unaccounted + second retained baseline. The producer's complete original capture remains + in its case for normal independent archival, even when that D7 case fails. + """ + + def __init__(self, root, *, reflink=True): + self.root = Path(root).absolute() + self.reflink = reflink + self.fd = None + + def __enter__(self): + require(self.fd is None, "reference store is already locked") + try: + self.root.mkdir(mode=0o700) + created = True + except FileExistsError: + created = False + private_directory(self.root) + self.fd = os.open(self.root / "lock", os.O_RDWR | os.O_CREAT | os.O_NOFOLLOW, 0o600) + try: + fcntl.flock(self.fd, fcntl.LOCK_EX | fcntl.LOCK_NB) + if created: + write_private(self.root / "store.json", seal({"schema": SCHEMA})) + (self.root / "retired").mkdir(mode=0o700) + sync(self.root) + sync(self.root.parent) + marker = sealed_file(self.root / "store.json") + require(marker == seal({"schema": SCHEMA}), "unrecognized reference store") + private_directory(self.root / "retired") + require( + {p.name for p in self.root.iterdir()} + <= {"lock", "store.json", "retired", "current"}, + "unfinished or unrecognized reference-store contents; evidence retained", + ) + return self + except BaseException: + os.close(self.fd) + self.fd = None + raise + + def __exit__(self, *_): + if self.fd is not None: + os.close(self.fd) + self.fd = None + + def _locked(self): + require(self.fd is not None, "reference store operation requires its lock") + require(not (self.root / "pending").exists(), "unfinished reference publication retained") + + def _key(self, identity, plan): + authenticate(identity) + require( + set(identity) == {"schema", "campaign", "configuration", "environment", "sha256"} + and identity["schema"] == SCHEMA + "/identity" + and isinstance(identity["configuration"], dict) + and bool(identity["configuration"]) + and isinstance(identity["environment"], dict) + and bool(identity["environment"]), + "reference identity requires campaign, effective configuration and environment", + ) + require_sha(identity["campaign"]) + compatible_serial_plans(plan, plan) + return digest({"identity": identity["sha256"], "plan": plan["sha256"]}) + + def _entry(self): + self._locked() + current = self.root / "current" + if not current.exists() and not current.is_symlink(): + return None + private_directory(current) + entry = sealed_file(current / "entry.json") + require(entry.get("schema") == SCHEMA + "/entry", "invalid reference entry") + require( + re.fullmatch(r"[0-9a-f]{32}", entry.get("publication_id", "")) is not None + and entry["identity"]["campaign"] == entry["campaign"], + "reference publication identity changed", + ) + require( + entry["key"] == self._key(entry["identity"], entry["plan"]), + "reference entry binding changed", + ) + projection = sealed_file(current / "capture/reference-projection.json") + schedule = sealed_file(current / "capture/schedule.json") + boundaries = sealed_file(current / "capture/boundaries/boundaries.json") + require( + projection["sha256"] == entry["projection"] + and projection["source_plan"] == projection["requested_plan"] == entry["plan"]["sha256"] + and projection["schedule"] == schedule["sha256"] + and projection["boundaries"] == boundaries["sha256"] + and bool(schedule["frames"]) + and bool(boundaries["frames"]), + "reference entry completion changed", + ) + return entry + + def project(self, identity, plan, requested, output): + """Return None for a different identity; corrupt matching entries raise.""" + key = self._key(identity, plan) + entry = self._entry() + if entry is None or entry["key"] != key: + return None + output = Path(output).absolute() + require(not output.resolve().is_relative_to(self.root.resolve()), "output is inside store") + result = project_serial_reference( + plan, requested, self.root / "current/capture", output, reflink=self.reflink + ) + write_private(output / "store-origin.json", entry) + return result + + def publish(self, identity, plan, source, origin): + """Clone a complete capture from a case, preserving its original there.""" + key = self._key(identity, plan) + require(self._entry() is None, "retire the previous reference before replacement") + source, origin = Path(source).resolve(strict=True), Path(origin).resolve(strict=True) + require( + re.fullmatch(r"case-\d{5}(?:-attempt-\d{3,})?", origin.name) is not None + and source.is_relative_to(origin) + and not source.is_relative_to(self.root.resolve()) + and not self.root.resolve().is_relative_to(origin), + "reference origin must be a separate qualification case", + ) + private_directory(origin) + payload = private_json(origin / "input.json") + authenticate(payload["campaign"]) + require(payload["case"] in payload["campaign"]["cases"], "origin case is not declared") + require( + identity["campaign"] == payload["campaign"]["sha256"], + "reference producer belongs to another campaign", + ) + origin_input = hashlib.sha256((origin / "input.json").read_bytes()).hexdigest() + pending = self.root / "pending" + pending.mkdir(mode=0o700) + # A complete full-domain projection authenticates every stored tensor. + projection = project_serial_reference( + plan, plan, source, pending / "capture", reflink=self.reflink + ) + entry = seal( + { + "schema": SCHEMA + "/entry", + "publication_id": uuid.uuid4().hex, + "key": key, + "identity": identity, + "plan": plan, + "origin": str(origin), + "capture_relative": str(source.relative_to(origin)), + "origin_input_sha256": origin_input, + "campaign": payload["campaign"]["sha256"], + "case": payload["case"], + "projection": projection["sha256"], + } + ) + write_private(pending / "entry.json", entry) + sync(pending) + pending.rename(self.root / "current") + sync(self.root) + return entry + + def _retained(self, entry): + origin = Path(entry["origin"]) + private_directory(origin) + private_json(origin / "input.json") + require( + hashlib.sha256((origin / "input.json").read_bytes()).hexdigest() + == entry["origin_input_sha256"], + "reference origin input changed", + ) + locator = origin / "archive-locator.json" + if locator.exists(): + archived = private_json(locator) + retired = private_json(origin / "archive-retirement.json") + result = sealed_file(origin / "result.json") + identity = archived.get("identity", {}) + require( + archived.get("schema") == "urn:qwen:conformance-archive-locator:v1" + and archived.get("archive_path") == f"/{origin.name}.tar" + and identity.get("scope") == "finished-case-v1" + and identity.get("input_sha256") == entry["origin_input_sha256"] + and identity.get("campaign") == result.get("campaign") == entry["campaign"] + and identity.get("result") == result["sha256"] + and result.get("case") == entry["case"] + and all(retired.get(k) == v for k, v in archived.items()) + and retired.get("metadata_retained") is True, + "reference archive retention is not established", + ) + for field in ("snapshot_id", "tar_sha256", "manifest_sha256"): + require_sha(archived[field]) + # The separate archive writer verifies every tar member before this + # receipt. This store does not claim a second remote archive readback. + return {"kind": "verified-case-archive", "locator": archived} + source = origin / entry["capture_relative"] + cached = self.root / "current/capture" + for relative in ("schedule.json", "boundaries/boundaries.json"): + schedule = sealed_file(cached / relative) + prefix = Path() if relative == "schedule.json" else Path("boundaries") + for frame in schedule["frames"]: + report = compare_frames( + source / prefix / frame["name"], cached / prefix / frame["name"] + ) + require(report["equal"], "retained original no longer matches cached reference") + return {"kind": "verified-original-capture", "capture": str(source)} + + def retire(self): + """Remove only this store's copy after establishing retained evidence.""" + entry = self._entry() + if entry is None: + return None + retained = self._retained(entry) + receipt = seal({"schema": SCHEMA + "/retirement", "entry": entry, "retained": retained}) + write_private(self.root / "retired" / (entry["sha256"] + ".json"), receipt) + sync(self.root / "retired") + shutil.rmtree(self.root / "current") + sync(self.root) + return receipt diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_replay.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_replay.py new file mode 100644 index 0000000..43f4f51 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_replay.py @@ -0,0 +1,416 @@ +"""Backend-neutral replay plans, independent prefill and operator capsules. + +Forced tokens are diagnostic inputs, never a replacement sampler or a repair +of the user's conversation. Only the separate checked session may publish. +""" + +from __future__ import annotations + +import hashlib +from pathlib import Path + +import numpy as np + +from qwen_r9700_lab.conformance_artifacts import reference_runtime_identity +from qwen_r9700_lab.conformance_model import Checkpoint, QuantizedQwenReference, state_names +from qwen_r9700_lab.conformance_reference import OPERATORS, reference_contract, reference_semantics +from qwen_r9700_lab.conformance_state import FrameWriter, load_arrays, read_frame +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + digest, + integer, + require_sha, + seal, + semantic_identity, + write_private, +) + +PLAN_SCHEMA = "urn:qwen:conformance-replay-plan:v1" +SCHEDULE_SCHEMA = "urn:qwen:conformance-schedule:v1" +OUTPUT_COMPONENTS = ["output.logits", "output.greedy"] + + +def reference_code_identity(): + identity = { + name: hashlib.sha256(Path(__file__).with_name(name + ".py").read_bytes()).hexdigest() + for name in ( + "conformance_model", + "conformance_reference", + "conformance_replay", + "conformance_state", + "conformance_boundaries", + "conformance_artifacts", + "diagnostic_contract", + "ordered_reference_linear", + "exact_fp8_metrics", + ) + } + identity["ordered_reference_linear.c"] = hashlib.sha256( + Path(__file__).with_name("ordered_reference_linear.c").read_bytes() + ).hexdigest() + return identity + + +def validate_plan(plan: dict) -> dict: + authenticate(plan) + if set(plan) - { + "accepted_widths", + "reference_profile", + "reference_semantics", + "reference_runtime", + "observation_positions", + "reference_linear", + } != { + "schema", + "contract", + "execution", + "adapter", + "checkpoint", + "checkpoint_files", + "kv_scales", + "prefix", + "forced_tokens", + "reference_arithmetic", + "sha256", + }: + raise DiagnosticError("replay plan has missing or unsupported fields") + if plan["schema"] != PLAN_SCHEMA: + raise DiagnosticError("unsupported replay plan") + for name in ("contract", "execution", "adapter"): + require_sha(plan[name]) + profile = plan.get("reference_profile", "radiance-fp8") + if plan["reference_arithmetic"] != reference_contract(profile): + raise DiagnosticError("reference arithmetic/runtime does not match this replay plan") + if ("reference_profile" in plan) != ("reference_semantics" in plan): + raise DiagnosticError("reference profile needs its complete semantic contract") + if "reference_semantics" in plan: + require_sha(plan.get("reference_runtime")) + semantics = plan["reference_semantics"] + expected = reference_semantics( + plan["checkpoint_files"], semantics["weights"]["config"], plan["kv_scales"], profile + ) + if semantics != expected or plan["contract"] != semantic_identity(expected): + raise DiagnosticError("plan settings differ from its reference contract") + if plan["adapter"] != digest(reference_code_identity()): + raise DiagnosticError("reference implementation changed; regenerate the replay plan") + if "reference_linear" in plan: + from qwen_r9700_lab.ordered_reference_linear import validate_binding + + validate_binding(plan["reference_linear"]) + for name in ("prefix", "forced_tokens"): + if not isinstance(plan[name], list) or not plan[name]: + raise DiagnosticError("initial prefill and forced-token schedule must be nonempty") + for token in plan[name]: + integer(token) + if token >= 2**31: + raise DiagnosticError("token outside portable representation") + widths = plan.get("accepted_widths", [0] * (len(plan["forced_tokens"]) - 1)) + if not isinstance(widths, list) or any(integer(k) > 7 for k in widths): + raise DiagnosticError("only explicit D7 accepted widths 0 through 7 are admitted") + if 1 + sum(k + 1 for k in widths) != len(plan["forced_tokens"]): + raise DiagnosticError("forced tokens do not cover the accepted-prefix schedule") + positions = plan.get("observation_positions") + if "observation_positions" in plan and ( + not isinstance(positions, list) + or not positions + or any( + type(p) is not int or p < 0 or p >= len(plan["prefix"]) + len(plan["forced_tokens"]) - 1 + for p in positions + ) + or positions != sorted(set(positions)) + ): + raise DiagnosticError( + "observation positions must be a nonempty ordered materialized subset" + ) + return plan + + +def scheduled_inputs(plan): + prefix = list(plan["prefix"]) + widths = plan.get("accepted_widths", [0] * (len(plan["forced_tokens"]) - 1)) + cursor = 0 + for index in range(len(widths) + 1): + if index: + advance = widths[index - 1] + 1 + prefix.extend(plan["forced_tokens"][cursor : cursor + advance]) + cursor += advance + pending = plan["forced_tokens"][cursor] + yield { + "name": f"frame-{index:06d}", + "phase": "prefill" if index == 0 else "step", + "consumed": len(prefix), + "input_digest": digest(prefix), + "pending": pending, + } + + +def observation_domain(plan): + """Every materialized token, including earlier prefill chunks and D7 rows. + + The emitted final token is pending, so must not appear in this domain. + Prefix identities use the same public token-ID digest as committed frames. + """ + tokens = plan["prefix"] + plan["forced_tokens"][:-1] + selected = plan.get("observation_positions", list(range(len(tokens)))) + admitted = frozenset(selected) + # Hash canonical JSON prefixes incrementally; avoid quadratic re-encoding + # for a 250K prompt. Hash copies add only the closing bracket. + stream, identities = hashlib.sha256(b"["), {} + for i, token in enumerate(tokens): + stream.update(("," if i else "").encode() + str(token).encode("ascii")) + if i in admitted: + prefix = stream.copy() + prefix.update(b"]") + identities[i] = prefix.hexdigest() + return selected, identities + + +def write_model_frame(model, path: Path, *, phase: str, pending: int, logits) -> dict: + writer = FrameWriter( + path, + contract=model.contract, + execution=model.execution, + adapter=model.adapter, + input_digest=digest(model.tokens), + phase=phase, + consumed=len(model.tokens), + pending=pending, + logical={"execution_mode": "forced_token_replay"}, + expected=state_names(model.config) + OUTPUT_COMPONENTS, + ) + writer.array("sequence.tokens", np.asarray(model.tokens, dtype="= len(self.expected): + raise DiagnosticError("backend produced an extra frame") + expected = self.expected[len(self.frames)] + frame = read_frame(path) + if path != self.root / expected["name"] or any( + frame[k] != expected[k] for k in expected if k != "name" + ): + raise DiagnosticError("backend did not follow the forced-token schedule") + if frame["coverage"] != self.coverage or frame["contract"] != self.plan["contract"]: + raise DiagnosticError("backend frame does not cover the reference contract") + if frame["logical"].get("execution_mode") != "forced_token_replay": + raise DiagnosticError("forced replay frame lacks its publication prohibition") + self.frames.append({**expected, "sha256": frame["sha256"]}) + + def finish(self): + if len(self.frames) != len(self.expected): + raise DiagnosticError("backend stopped before the replay finished") + result = seal( + { + "schema": SCHEDULE_SCHEMA, + "contract": self.plan["contract"], + "plan": self.plan["sha256"], + "frames": self.frames, + "coverage": self.coverage, + "backend": self.backend, + "initial_state": "independent_zero_state", + "published_to_session": False, + "native_equivalence": "UNPROVED", + } + ) + write_private(self.root / "schedule.json", result) + return result + + +def run_reference(plan: dict, root: Path): + validate_plan(plan) + if ( + "reference_runtime" in plan + and plan["reference_runtime"] != reference_runtime_identity()["sha256"] + ): + raise DiagnosticError( + "CPU reference executable/runtime changed; regenerate the replay plan" + ) + checkpoint = Checkpoint(Path(plan["checkpoint"]), plan["checkpoint_files"]) + model = QuantizedQwenReference( + checkpoint, + kv_scales=plan["kv_scales"], + contract=plan["contract"], + execution=plan["execution"], + adapter=plan["adapter"], + reference_profile=plan.get("reference_profile", "radiance-fp8"), + ) + try: + accelerator = None + if "reference_linear" in plan: + from qwen_r9700_lab import conformance_reference + from qwen_r9700_lab.ordered_reference_linear import OrderedLinear + + accelerator = OrderedLinear(conformance_reference, plan["reference_linear"]) + model._linear = accelerator + if ( + "reference_semantics" in plan + and model.config != plan["reference_semantics"]["weights"]["config"] + ): + raise DiagnosticError("checkpoint configuration differs from the reference contract") + campaign = CampaignWriter( + root, + plan, + coverage=state_names(model.config) + OUTPUT_COMPONENTS, + backend=( + "independent-ordered-c-numpy-reference" + if accelerator is not None + else "independent-numpy-reference" + ), + ) + from qwen_r9700_lab.conformance_boundaries import BoundaryRecorder, detailed_stages + + positions, inputs = observation_domain(plan) + boundaries = BoundaryRecorder( + root / "boundaries", + contract=model.contract, + execution=model.execution, + adapter=model.adapter, + positions=positions, + layers=model.config["num_hidden_layers"], + input_digests=inputs, + ) + detailed = BoundaryRecorder( + root / "semantic", + contract=model.contract, + execution=model.execution, + adapter=model.adapter, + positions=boundaries.positions, + layers=model.config["num_hidden_layers"], + input_digests=boundaries.inputs, + layer_stages=detailed_stages(model.config), + ) + + def observe(position, layer, stage, value): + boundaries.record(position, layer, stage, value) + detailed.record(position, layer, stage, value) + + model.capture = observe + logits = None + for token in plan["prefix"]: + logits = model.step(token) + for index, expected in enumerate(campaign.expected): + if index: + begin = len(model.tokens) - len(plan["prefix"]) + end = expected["consumed"] - len(plan["prefix"]) + for token in plan["forced_tokens"][begin:end]: + logits = model.step(token) + path = root / expected["name"] + write_model_frame( + model, path, phase=expected["phase"], pending=expected["pending"], logits=logits + ) + campaign.record(path) + boundaries.finish() + detailed.finish() + if accelerator is not None: + write_private( + root / "ordered-linear.json", + seal( + { + "binding": accelerator.binding, + "calls": accelerator.calls, + "fallbacks": accelerator.fallbacks, + "formal_equivalence": "UNPROVED", + } + ), + ) + return campaign.finish() + finally: + model.close() + + +def write_operator_capsule( + root: Path, + *, + operator: str, + inputs: dict, + outputs: list, + options: dict, + contract: str, + execution: str, + adapter: str, +): + if operator not in OPERATORS or not inputs or not outputs: + raise DiagnosticError("operator capsule has no supported comparison domain") + inputs = dict(sorted(inputs.items())) + names = ["input." + k for k in inputs] + [f"output.{i}" for i in range(len(outputs))] + identity = digest( + { + "operator": operator, + "options": options, + "inputs": { + k: hashlib.sha256(np.ascontiguousarray(v).tobytes()).hexdigest() + for k, v in inputs.items() + }, + } + ) + writer = FrameWriter( + root, + contract=contract, + execution=execution, + adapter=adapter, + input_digest=identity, + phase="operator", + consumed=0, + pending=None, + expected=names, + logical={"operator": operator, "options": options}, + ) + for name, value in inputs.items(): + writer.array("input." + name, np.asarray(value)) + for index, value in enumerate(outputs): + writer.array(f"output.{index}", np.asarray(value)) + return writer.finish() + + +def replay_operator(capsule: Path, output: Path): + from qwen_r9700_lab.conformance_state import compare_frames + + frame, arrays = load_arrays(capsule) + operator = frame["logical"].get("operator") + if frame["phase"] != "operator" or operator not in OPERATORS: + raise DiagnosticError("unsupported operator capsule") + kwargs = {k.removeprefix("input."): v for k, v in arrays.items() if k.startswith("input.")} + options = frame["logical"]["options"] + if set(kwargs) & set(options): + raise DiagnosticError("operator options overwrite captured inputs") + result = OPERATORS[operator](**kwargs, **options) + outputs = list(result) if isinstance(result, tuple) else [result] + write_operator_capsule( + output, + operator=operator, + inputs=kwargs, + outputs=outputs, + options=options, + contract=frame["contract"], + execution=frame["execution"], + adapter=frame["adapter"], + ) + return compare_frames(output, capsule) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_rotary_intervention.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_rotary_intervention.py new file mode 100644 index 0000000..def06a3 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_rotary_intervention.py @@ -0,0 +1,121 @@ +"""Admit only the declared native RoPE rounding change in a saved eager replay.""" + +from copy import deepcopy + +from qwen_r9700_lab.conformance_execution_modes import admit_pair, compare_pair +from qwen_r9700_lab.conformance_precision_intervention import ( + admit_precision_intervention, + compare_precision_intervention, +) +from qwen_r9700_lab.conformance_topk import require +from qwen_r9700_lab.diagnostic_contract import authenticate, seal + + +def normalized_rotary(reference, candidate, binding): + authenticate(binding) + for side in (reference, candidate): + for key in ("measurement", "config", "runtime", "pass"): + authenticate(side[key]) + require(side["measurement"]["execution_mode"] == "eager", "rotary arm must be eager") + require( + reference["measurement"]["driver_sha256"] == binding["base_driver"], "base driver changed" + ) + require( + candidate["measurement"]["driver_sha256"] == binding["experiment_driver"], + "rotary experiment driver changed", + ) + capture = candidate["measurement"]["isolated_capture"] + require(reference["measurement"]["isolated_capture"] == capture, "capture controls differ") + require( + reference["config"]["worker_cls"] + == "execution_mode_d7_worker.ExecutionMode" + ("CaptureWorker" if capture else "Worker"), + "unexpected reference worker", + ) + require( + candidate["config"]["worker_cls"] + == "rotary_mode_d7_worker.RotaryRne" + ("CaptureWorker" if capture else "Worker"), + "unexpected rotary worker", + ) + before = candidate["runtime"]["rotary_intervention"] + after = candidate["pass"]["observation"]["rotary_intervention"] + for observation in (before, after): + identity = observation["identity"] + authenticate(identity) + require(identity["installed_before_load"] is True, "late rotary installation") + for key in ("native_source", "patched_source", "worker_source", "patcher_source"): + require(identity[key] == binding[key], f"rotary intervention changes {key}") + require(before["identity"] == after["identity"], "rotary identity changed during replay") + require(after["calls"] > before["calls"] >= 0, "modified rotary did not execute during replay") + normalized = deepcopy(candidate) + normalized["measurement"].pop("sha256") + normalized["measurement"]["driver_sha256"] = binding["base_driver"] + normalized["measurement"] = seal(normalized["measurement"]) + # All other config/runtime/source checks still run in ordinary admission. + admit_pair(reference, normalized) + return normalized + + +def compare_rotary_intervention(reference, candidate, reference_rows, candidate_rows, binding): + normalized = normalized_rotary(reference, candidate, binding) + result = compare_pair(reference, normalized, reference_rows, candidate_rows) + return seal( + { + "schema": "qwen.rotary-intervention-comparison.v1", + "status": "COMPARED_DECLARED_INTERVENTION", + "binding": binding, + "normalized_comparison": result, + "original_receipts": [ + [s[k]["sha256"] for k in ("measurement", "config", "runtime", "pass")] + for s in (reference, candidate) + ], + "observed_rotary": candidate["pass"]["observation"]["rotary_intervention"], + "decode": result["decode"], + "prefill": result["prefill"], + "scope": ( + "Same eager replay, one native product-rounding intervention; sampled outputs only." + ), + } + ) + + +def compare_common_rounding(reference, candidate, compiled, candidate_rows, compiled_rows, binding): + normalized = normalized_rotary(reference, candidate, binding) + result = compare_precision_intervention(normalized, compiled, candidate_rows, compiled_rows) + return seal( + { + "schema": "qwen.rotary-and-casts-comparison.v1", + "status": "COMPARED_TWO_DECLARED_INTERVENTIONS", + "binding": binding, + "normalized_comparison": result, + "original_receipts": [ + [s[k]["sha256"] for k in ("measurement", "config", "runtime", "pass")] + for s in (reference, candidate, compiled) + ], + "observed_rotary": candidate["pass"]["observation"]["rotary_intervention"], + "decode": result["decode"], + "prefill": result["prefill"], + "scope": ( + "Eager native RoPE products changed to nearest-even; compiled precision-cast " + "emulation enabled. Other metadata admitted unchanged; 320 sampled predictions." + ), + } + ) + + +def admit_common_rounding(reference, candidate, compiled, binding): + normalized = normalized_rotary(reference, candidate, binding) + admission = admit_precision_intervention(normalized, compiled) + return seal( + { + "schema": "qwen.rotary-and-casts-admission.v1", + "status": "ADMITTED_TWO_DECLARED_INTERVENTIONS", + "binding": binding, + "normalized_admission": admission, + "original_receipts": [ + [s[k]["sha256"] for k in ("measurement", "config", "runtime", "pass")] + for s in (reference, candidate, compiled) + ], + "captures": admission["captures"], + "scope": "RNE RoPE and compiler cast preservation; other controls checked.", + } + ) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_rotary_repair.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_rotary_repair.py new file mode 100644 index 0000000..a3090b5 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_rotary_repair.py @@ -0,0 +1,40 @@ +"""Construct an isolated, pinned MRoPE product-rounding intervention. + +This never installs a production override. A native probe must validate the +generated module against the recorded input/output contract before using it. +""" + +import hashlib + +from qwen_r9700_lab.conformance_topk import require + +SOURCE_SHA256 = "f795e2a347715d1c8e2b9953fcf6fba578ca9a648858ed9d26d27eb6bdd61d19" + +HELPER = """@triton.jit +def _diagnostic_bf16_product_rne(a, b): + return (a.to(tl.float32) * b.to(tl.float32)).to( + a.dtype, fp_downcast_rounding="rtne" + ) + + +""" + + +def patch_source(source): + require( + hashlib.sha256(source.encode()).hexdigest() == SOURCE_SHA256, + "native MRoPE source is outside the pinned intervention", + ) + marker = "@triton.jit\ndef _triton_mrope_forward(" + require(source.count(marker) == 1, "missing unique native rotary entry") + result = source.replace(marker, HELPER + marker) + for kind in ("q", "k"): + for index in (1, 2): + for coefficient in ("cos_row", "sin_row"): + expression = f"{kind}_tile_{index} * {coefficient}" + require(result.count(expression) == 2, "native rotary arithmetic changed") + result = result.replace( + expression, + f"_diagnostic_bf16_product_rne({kind}_tile_{index}, {coefficient})", + ) + return result diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_runtime.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_runtime.py new file mode 100644 index 0000000..d508657 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_runtime.py @@ -0,0 +1,350 @@ +"""Launch a pinned, separate vLLM API process for production-path qualification. + +Run in the prepared Radiance Python environment/container. This module never +builds, installs, patches or restarts the production service. All mutable server +paths and IPC metadata are redirected to this campaign's new private directory. +""" + +from __future__ import annotations + +import copy +import hashlib +import json +import os +import secrets +import shutil +import socket +import sys +from pathlib import Path + +from qwen_r9700_lab.conformance_shm import OwnedOffloadRegion +from qwen_r9700_lab.conformance_transport import OwnedClient, OwnedProcess +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, digest, private_json, write_private + +MODULES = ( + "vllm.v1.worker.gpu.model_runner", + "qwen_radiance_fair_scheduler", + "qwen_radiance_chat_tier", + "qwen_radiance_cache", + "radiance_verifyhead", + "vllm.parser.qwen3", + "vllm.parser.engine.parser_engine", + "vllm.parser.engine.adapters", + "vllm.parser.parser_manager", +) + +SHUTDOWN_TIMEOUT_SECONDS = 60 +SHUTDOWN_CLEANUP_SECONDS = 15 + + +def isolated_config(config, root, nonce, port, *, graphs, speculation, asynchronous=False): + config = copy.deepcopy(config) + shutdown = config.setdefault("shutdown_timeout", SHUTDOWN_TIMEOUT_SECONDS) + if isinstance(shutdown, bool) or not isinstance(shutdown, int) or shutdown <= 0: + raise DiagnosticError("persistent-cache qualification requires a positive shutdown timeout") + if not isinstance(config.get("model"), str) or not Path(config["model"]).is_absolute(): + raise DiagnosticError("qualification requires an explicit local checkpoint") + if config.get("tensor_parallel_size", 1) != 1 or config.get("pipeline_parallel_size", 1) != 1: + raise DiagnosticError("the qualification server admits TP1/PP1 only") + if config.get("data_parallel_size", 1) != 1: + raise DiagnosticError("distributed qualification is not implemented") + config.update( + host="127.0.0.1", + port=port, + api_key=nonce, + max_num_seqs=2, + async_scheduling=asynchronous, + enforce_eager=not graphs, + enable_prefix_caching=True, + mamba_cache_mode="align", + middleware=[ + "qwen_r9700_lab.conformance_runtime.identity_middleware", + "qwen_radiance_request_guard.require_snapshot_abi", + ], + ) + if not speculation: + config.pop("speculative_config", None) + elif not isinstance(config.get("speculative_config"), dict): + raise DiagnosticError("natural speculation requires the pinned drafter configuration") + if graphs: + config["compilation_config"] = { + "cudagraph_mode": "PIECEWISE", + "cudagraph_capture_sizes": [1, 2, 4, 8], + } + else: + config.pop("compilation_config", None) + # Async vLLM is a different, explicitly experimental scheduler domain. The + # production FairScheduler rejects it; never imply that it was exercised. + if asynchronous: + config.pop("scheduler_cls", None) + config.pop("additional_config", None) + config.pop("kv_transfer_config", None) + else: + config["scheduler_cls"] = "qwen_radiance_fair_scheduler.FairScheduler" + additional = config.setdefault("additional_config", {}) + fair = additional.setdefault("qwen_fair", {}) + fair.update(status_path=str(root / "fair"), max_cached_chats=2, policy="response_boundary") + transfer = config.get("kv_transfer_config") + if not isinstance(transfer, dict) or transfer.get("kv_connector") != "OffloadingConnector": + raise DiagnosticError("production qualification requires the real snapshot connector") + transfer.update( + engine_id="conformance-" + nonce, kv_role="kv_both", kv_load_failure_policy="fail" + ) + extra = transfer["kv_connector_extra_config"] + tiers = extra.get("secondary_tiers", []) + if len(tiers) != 1 or tiers[0].get("type") != "qwen_chat_fs": + raise DiagnosticError("unreviewed qualification snapshot tier") + tiers[0].update( + root_dir=str(root / "data"), + control_directory=str(root / "control"), + tail_status_path=str(root / "tail.json"), + ) + return config + + +def server_argv(config): + argv = ["vllm.entrypoints.openai.api_server"] + for key, value in config.items(): + if not key.replace("_", "").isalnum() or not key[0].isalpha(): + raise DiagnosticError("invalid server argument name") + flag = "--" + key.replace("_", "-") + if isinstance(value, bool): + # vLLM's BooleanOptionalAction is used for async scheduling. Other + # disabled switches must simply be omitted (e.g. enforce_eager). + if value: + argv.append(flag) + elif key == "async_scheduling": + argv.append("--no-async-scheduling") + elif value is None: + continue + elif isinstance(value, dict): + argv.extend((flag, json.dumps(value, separators=(",", ":")))) + elif isinstance(value, list): + if key == "middleware": + for item in value: + argv.extend((flag, str(item))) + else: + argv.extend((flag, *(str(item) for item in value))) + else: + argv.extend((flag, str(value))) + return argv + + +def worker_environment(spec, root): + env = dict(os.environ) + env.update(spec["environment"]) + for key, directory in { + "VLLM_CACHE_ROOT": "vllm", + "TORCHINDUCTOR_CACHE_DIR": "inductor", + "TRITON_CACHE_DIR": "triton", + "XDG_CACHE_HOME": "xdg-cache", + "TORCH_EXTENSIONS_DIR": "torch-extensions", + "CUDA_CACHE_PATH": "cuda", + }.items(): + path = root / "runtime" / directory + path.mkdir(parents=True, mode=0o700, exist_ok=True) + env[key] = str(path) + # AITER_ROOT_DIR includes its source/JIT tree, not just an empty cache. + # Clone a configured tree before allowing the qualifier to compile in it. + if env.get("AITER_ROOT_DIR"): + aiter_source = Path(env["AITER_ROOT_DIR"]).resolve() + aiter_copy = root / "runtime/aiter" + if not aiter_copy.exists(): + shutil.copytree(aiter_source, aiter_copy) + env["AITER_ROOT_DIR"] = str(aiter_copy) + env.update( + QWEN_CONFORMANCE_GPU="1", + QWEN_RADIANCE_CACHE_ABI=spec["binding"]["live_data_abi"], + HF_HUB_OFFLINE="1", + TRANSFORMERS_OFFLINE="1", + VLLM_USE_V2_MODEL_RUNNER="1", + ) + # Source is explicitly bound in the campaign. Do not rely on whichever + # editable installation an unrelated shell happens to have activated. + source = str(Path(__file__).resolve().parents[1]) + env["PYTHONPATH"] = source + os.pathsep + env.get("PYTHONPATH", "") + return env + + +class NativeServer: + def __init__( + self, + spec, + root, + *, + allow_gpu=False, + graphs=True, + speculation=True, + head=True, + head_audit=False, + asynchronous=False, + observe=True, + dynamic_width=False, + head_fault=False, + primary_cache_bytes=None, + ): + if not allow_gpu: + raise DiagnosticError("GPU use was not authorized") + interpreter = Path(spec["python"]) + if hashlib.sha256(interpreter.read_bytes()).hexdigest() != spec["python_sha256"]: + raise DiagnosticError("qualification Python artifact changed") + self.spec, self.root = spec, Path(root).resolve() + self.root.mkdir(mode=0o700) + self.variant = { + "graphs": graphs, + "speculation": speculation, + "head": head, + "head_audit": head_audit, + "asynchronous": asynchronous, + "observe": observe, + "dynamic_width": dynamic_width, + "head_fault": head_fault, + "primary_cache_bytes": primary_cache_bytes, + } + self.process, self.client, self.incarnation = None, None, 0 + self.offload_region = None + self.offload_receipt = None + self.executions = [] + self.shutdown_grace_seconds = SHUTDOWN_TIMEOUT_SECONDS + SHUTDOWN_CLEANUP_SECONDS + + def start(self): + if self.process is not None: + raise DiagnosticError("qualification server is already running") + nonce = secrets.token_hex(32) + reservation = socket.socket() + reservation.bind(("127.0.0.1", 0)) + port = reservation.getsockname()[1] + config = isolated_config( + self.spec["server_config"], + self.root, + nonce, + port, + **{k: self.variant[k] for k in ("graphs", "speculation", "asynchronous")}, + ) + self.shutdown_grace_seconds = config["shutdown_timeout"] + SHUTDOWN_CLEANUP_SECONDS + capacity = self.variant["primary_cache_bytes"] + if capacity is not None: + if isinstance(capacity, bool) or not isinstance(capacity, int) or capacity <= 0: + reservation.close() + raise DiagnosticError("invalid qualification primary-cache capacity") + if self.variant["asynchronous"]: + reservation.close() + raise DiagnosticError("async qualification has no primary snapshot cache") + extra = config["kv_transfer_config"]["kv_connector_extra_config"] + original = extra["cpu_bytes_to_use"] + if capacity >= original: + reservation.close() + raise DiagnosticError("eviction qualification must reduce primary-cache capacity") + extra["cpu_bytes_to_use"] = capacity + sources = self.spec["observer_sources"] if self.variant["observe"] else {} + if self.variant["observe"] and set(sources) != set(MODULES): + reservation.close() + raise DiagnosticError("qualification observer source inventory is incomplete") + settings = { + "root": str(self.root), + "nonce": nonce, + "config": config, + "observer_sources": sources, + "head_audit": self.variant["head_audit"], + "head_fault": self.variant["head_fault"], + "binding": self.spec["binding"], + "variant": self.variant, + "incarnation": self.incarnation, + } + settings["execution"] = digest(settings) + settings_path = self.root / f"server-{self.incarnation}.json" + write_private(settings_path, settings) + env = worker_environment(self.spec, self.root) + env.update( + QWEN_CONFORMANCE_SERVER_SETTINGS=str(settings_path), + RADIANCE_VERIFY_HEAD="1" if self.variant["head"] else "0", + RADIANCE_DYNAMIC_WIDTH="1" if self.variant["dynamic_width"] else "0", + ) + overlay = self.root / f"overlay-{self.incarnation}" + overlay.mkdir(mode=0o700) + if self.variant["observe"]: + # Python normally swallows sitecustomize exceptions. Exit explicitly + # on a hook/bootstrap failure so an unobserved run cannot proceed. + code = ( + "import os, traceback\ntry:\n" + " from qwen_r9700_lab.conformance_observer import install_from_environment\n" + " install_from_environment()\nexcept BaseException:\n" + " traceback.print_exc()\n os._exit(87)\n" + ) + (overlay / "sitecustomize.py").write_text(code) + env["PYTHONPATH"] = str(overlay) + os.pathsep + env["PYTHONPATH"] + argv = [self.spec["python"], "-m", "qwen_r9700_lab.conformance_runtime", str(settings_path)] + reservation.close() + try: + transfer = config.get("kv_transfer_config") + if transfer is not None: + self.offload_region = OwnedOffloadRegion(transfer["engine_id"]) + self.offload_receipt = self.root / f"shared-memory-{self.incarnation}.json" + self.process = OwnedProcess( + argv, + self.root / f"process-{self.incarnation}", + env=env, + timeout=self.spec["case_timeout_seconds"], + ) + self.client = OwnedClient(self.process, port, nonce, settings["execution"]) + self.client.connect(timeout=self.spec["startup_timeout_seconds"]) + except BaseException: + self.stop() + raise + self.settings = settings + self.executions.append(settings["execution"]) + self.incarnation += 1 + return self + + def stop(self, *, crash=False): + if self.process is not None: + self.process.close(crash=crash, grace_seconds=self.shutdown_grace_seconds) + self.process = None + self.client = None + if self.offload_region is not None: + region, self.offload_region = self.offload_region, None + write_private(self.offload_receipt, region.release()) + + def restart(self, *, crash=False): + self.stop(crash=crash) + return self.start() + + def __enter__(self): + return self.start() + + def __exit__(self, *_): + self.stop() + + +async def identity_middleware(request, call_next): + if request.url.path != "/qwen-conformance/identity": + return await call_next(request) + from starlette.responses import JSONResponse + + settings = private_json(Path(os.environ["QWEN_CONFORMANCE_SERVER_SETTINGS"])) + if request.headers.get("authorization") != "Bearer " + settings["nonce"]: + return JSONResponse({"error": "unauthorized qualification client"}, status_code=403) + return JSONResponse( + {"nonce": settings["nonce"], "execution": settings["execution"], "pid": os.getpid()} + ) + + +def main(): + if os.environ.get("QWEN_CONFORMANCE_GPU") != "1": + raise DiagnosticError("qualification server is not armed") + settings = private_json(Path(sys.argv[1])) + import importlib.util + import runpy + + from qwen_r9700_lab.conformance_radiance import verify_sources + + spec = importlib.util.find_spec("vllm") + if spec is None or spec.origin is None: + raise DiagnosticError("pinned Radiance package is not installed") + verify_sources(Path(spec.origin).parent.parent, settings["binding"]) + sys.argv = server_argv(settings["config"]) + runpy.run_module("vllm.entrypoints.openai.api_server", run_name="__main__") + + +if __name__ == "__main__": + main() diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_scenarios.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_scenarios.py new file mode 100644 index 0000000..c700a72 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_scenarios.py @@ -0,0 +1,1528 @@ +"""Concrete native and production-interface campaign drivers. + +All prompts are deterministic synthetic records. No saved Pi transcript, user's +cache directory or pre-existing server endpoint is accepted by these drivers. +""" + +from __future__ import annotations + +import contextlib +import hashlib +import json +import os +import random +import time +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +from qwen_r9700_lab.conformance_boundaries import compare_boundaries +from qwen_r9700_lab.conformance_observer import read_events +from qwen_r9700_lab.conformance_runtime import NativeServer, worker_environment +from qwen_r9700_lab.conformance_session import compare_campaign +from qwen_r9700_lab.conformance_state import compare_frames, read_frame +from qwen_r9700_lab.conformance_transport import BackendResponseError, OwnedProcess +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + digest, + private_json, + seal, + write_private, +) +from qwen_r9700_lab.radiance_cache import ChatStore, cache_salt, request_tail_flush + + +class UnavailableError(DiagnosticError): + pass + + +def require(condition, message): + if not condition: + raise DiagnosticError(message) + + +def synthetic_text(seed, lines): + rng = random.Random(seed) + # Nonrepeating public records, not the repeated prose that can inflate + # speculative acceptance and conceal repetition failures in a benchmark. + return "\n".join( + json.dumps( + { + "record": i, + "amount": rng.randrange(1, 10000000), + "tag": f"{rng.getrandbits(64):016x}", + }, + sort_keys=True, + ) + for i in range(lines) + ) + + +def tokens(spec, count, seed): + from tokenizers import Tokenizer + + checkpoint = Path(spec["native_config"]["model"]) + tokenizer = Tokenizer.from_file(str(checkpoint / "tokenizer.json")) + result = tokenizer.encode(synthetic_text(seed, count // 8 + 32), add_special_tokens=False).ids + require(len(result) >= count, "synthetic fixture did not fill the declared token window") + return result[:count] + + +def plan_for(spec, case, forced, *, accepted=None): + from qwen_r9700_lab.conformance_cli import make_plan + + prefix = tokens(spec, case["context"], case["seed"]) + # Explicit capture windows retain boundary-adjacent and accepted rows. This + # limits diagnostic tensor traffic; the omitted positions remain unobserved. + positions = sorted( + {0, len(prefix) - 1, *list(range(max(0, len(prefix) - 8), len(prefix) + len(forced) - 1))} + ) + plan, _, _ = make_plan( + { + "checkpoint": spec["native_config"]["model"], + "checkpoint_files": spec["checkpoint_files"], + "kv_scales": spec["kv_scales"], + "reference_profile": spec["reference_profile"], + "prefix": prefix, + "forced_tokens": forced, + "observation_positions": positions, + **({"accepted_widths": accepted} if accepted is not None else {}), + **( + {"reference_linear": spec["reference_linear"]} if "reference_linear" in spec else {} + ), + } + ) + return plan + + +def native_configuration(spec, *, speculation=True): + config = dict(spec["native_config"]) + config.update(enforce_eager=True, max_num_seqs=1, async_scheduling=False) + config.pop("kv_transfer_config", None) + config.pop("scheduler_cls", None) + config.pop("additional_config", None) + config.pop("compilation_config", None) + if not speculation: + config.pop("speculative_config", None) + return config + + +def native(spec, plan, root, *, speculation=True, experiment=None): + from qwen_r9700_lab.conformance_cli import run_native + + root.mkdir(mode=0o700) + config = native_configuration(spec, speculation=speculation) + write_private(root / "plan.json", plan) + write_private(root / "config.json", config) + write_private(root / "binding.json", spec["binding"]) + prior = { + key: os.environ.get(key) + for key in ( + "RADIANCE_VERIFY_HEAD", + "QWEN_CONFORMANCE_NATIVE_EXPERIMENT", + "QWEN_CONFORMANCE_CALL_MODE", + ) + } + os.environ["RADIANCE_VERIFY_HEAD"] = "0" + os.environ["QWEN_CONFORMANCE_CALL_MODE"] = spec["native_call_mode"] + os.environ.pop("QWEN_CONFORMANCE_NATIVE_EXPERIMENT", None) + if experiment is not None: + write_private(root / "experiment.json", experiment) + os.environ["QWEN_CONFORMANCE_NATIVE_EXPERIMENT"] = str(root / "experiment.json") + try: + run_native( + root / "plan.json", + root / "config.json", + root / "binding.json", + root / "run", + allow_gpu=True, + ) + finally: + for key, value in prior.items(): + if value is None: + os.environ.pop(key, None) + else: + os.environ[key] = value + return root / "run/capture" + + +def serial_reference_identity(spec, root): + """Bind all effective settings, hashing environment values instead of logging them.""" + from qwen_r9700_lab.conformance_reference_store import SCHEMA, copied_tree_identity + + payload = private_json(root / "input.json") + campaign = payload["campaign"] + authenticate(campaign) + require(campaign["spec"] == spec, "serial reference spec differs from its case") + env = dict(os.environ) + env.update(RADIANCE_VERIFY_HEAD="0", QWEN_CONFORMANCE_CALL_MODE=spec["native_call_mode"]) + env.pop("QWEN_CONFORMANCE_NATIVE_EXPERIMENT", None) + # Only exact per-case scratch paths installed by worker_environment may be + # normalized. Unknown variables/paths remain bound, causing a safe miss. + for key, directory in { + "VLLM_CACHE_ROOT": "vllm", + "TORCHINDUCTOR_CACHE_DIR": "inductor", + "TRITON_CACHE_DIR": "triton", + "XDG_CACHE_HOME": "xdg-cache", + "TORCH_EXTENSIONS_DIR": "torch-extensions", + "CUDA_CACHE_PATH": "cuda", + }.items(): + if env.get(key) == str(root / "runtime" / directory): + env[key] = "/" + directory + # worker_environment copies this source/JIT tree, unlike empty scratch + # caches. Bind every copied byte before execution; never normalize away a + # different kernel tree just because its directory name is familiar. + aiter_tree = root / "runtime/aiter" + if env.get("AITER_ROOT_DIR") == str(aiter_tree): + env["AITER_ROOT_DIR"] = { + "path": "/aiter", + "copied_tree": copied_tree_identity(aiter_tree), + } + return seal( + { + "schema": SCHEMA + "/identity", + "campaign": campaign["sha256"], + "configuration": native_configuration(spec, speculation=False), + "environment": {k: digest(v) for k, v in sorted(env.items())}, + } + ) + + +def serial_reference_for_d7(spec, case, requested, root): + """Reuse one complete serial eleven-token baseline; each D7 arm stays fresh.""" + from qwen_r9700_lab.conformance_reference_store import SerialReferenceStore + + require(private_json(root / "input.json")["case"] == case, "serial reference case changed") + complete = plan_for(spec, case, tokens(spec, 11, case["seed"] + 1701)) + identity = serial_reference_identity(spec, root) + serial = root / "serial" + serial.mkdir(mode=0o700) + write_private(serial / "plan.json", requested) + write_private(serial / "config.json", identity["configuration"]) + write_private(serial / "binding.json", spec["binding"]) + (serial / "run").mkdir(mode=0o700) + destination = serial / "run/capture" + with SerialReferenceStore(root.parent / "serial-reference-store") as store: + projection = store.project(identity, complete, requested, destination) + hit = projection is not None + if not hit: + store.retire() + baseline = native(spec, complete, root / "serial-population", speculation=False) + store.publish(identity, complete, baseline, root) + projection = store.project(identity, complete, requested, destination) + require(projection is not None, "new serial reference was not published") + write_private( + serial / "reuse.json", + seal( + { + "identity": identity["sha256"], + "hit": hit, + "projection": projection["sha256"], + "fresh_d7_required": True, + } + ), + ) + return destination + + +def aligned_frames(reference, candidate, root, *, require_equal=True): + schedules = [private_json(p / "schedule.json") for p in (reference, candidate)] + for schedule in schedules: + authenticate(schedule) + require( + schedule.get("initial_state") == "independent_zero_state", + "native comparison reused an unvalidated initial state", + ) + reference_frames = {(f["consumed"], f["pending"]): f for f in schedules[0]["frames"]} + reports = [] + for frame in schedules[1]["frames"]: + corresponding = reference_frames.get((frame["consumed"], frame["pending"])) + require(corresponding is not None, "M1 replay missed a D7 committed position") + result = compare_frames(reference / corresponding["name"], candidate / frame["name"]) + reports.append(result) + write_private(root / "aligned-comparison.json", {"comparisons": reports}) + require(bool(reports), "M1/D7 comparison has no aligned frames") + if require_equal: + require( + all(r["equal"] for r in reports), + "M1/D7 logical state or logits diverged; comparison retained", + ) + return reports + + +def forced_reference(spec, case, root): + from qwen_r9700_lab.conformance_gpu_lease import gpu_lease + from qwen_r9700_lab.conformance_replay import run_reference + + forced = tokens(spec, 4, case["seed"] + 1701) + plan = plan_for(spec, case, forced) + run_reference(plan, root / "reference") + with gpu_lease(root / "gpu-lease"): + candidate = native(spec, plan, root / "candidate", speculation=False) + state = compare_campaign(root / "reference", candidate, root / "state-comparison") + boundary = compare_boundaries( + root / "reference/boundaries", candidate / "boundaries", root / "boundary-comparison" + ) + require( + state["equal"] and boundary["equal"], + "independent reference diverged; first boundaries retained", + ) + return [ + "independent_initial_prefill", + "forced_transition_state_and_logits", + "selected_layer_boundaries", + ] + + +def forced_d7(spec, case, root): + width = case["axes"]["accepted"] + forced = tokens(spec, width + 4, case["seed"] + 1701) + plan = plan_for(spec, case, forced, accepted=[width, 0, 0]) + serial_plan = plan_for(spec, case, forced) + serial = ( + serial_reference_for_d7(spec, case, serial_plan, root) + if spec.get("reuse_serial_reference", False) + else native(spec, serial_plan, root / "serial", speculation=False) + ) + candidate = native(spec, plan, root / "d7") + state = aligned_frames(serial, candidate, root, require_equal=False) + boundary = compare_boundaries( + serial / "boundaries", candidate / "boundaries", root / "boundary-comparison" + ) + # Complete both diagnoses before failing the case. A state mismatch is a + # reason to retain the causal boundary report, not a reason to omit it. + require( + all(r["equal"] for r in state), + "M1/D7 logical state or logits diverged; comparison retained", + ) + require(boundary["equal"], "M1/D7 causal layer boundary divergence") + return [ + f"accept_width_{width}", + "same_consumed_pending_and_all_logical_state", + "selected_causal_rows", + "initial_prefill", + ] + + +def rejected_suffix(spec, case, root): + width = case["axes"]["accepted"] + forced = tokens(spec, width + 4, case["seed"] + 1701) + plan = plan_for(spec, case, forced, accepted=[width, 0, 0]) + a = native(spec, plan, root / "suffix-a", experiment={"rejected_suffix_token": 0}) + b = native(spec, plan, root / "suffix-b", experiment={"rejected_suffix_token": 1}) + report = compare_campaign(a, b, root / "comparison") + require(report["equal"], "rejected draft suffix contaminated retained state or logits") + return ["accepted_prefix_fixed", "rejected_suffix_changed", "exact_committed_state_and_logits"] + + +def native_fault(spec, case, root): + fault = case["variant"] + plan = plan_for(spec, case, tokens(spec, 3, 1701)) + clean = native(spec, plan, root / "clean", speculation=False) + # A no-fault repeated capture first establishes that this negative control + # is not just counting an unrelated natural numerical divergence as a hit. + repeat = native(spec, plan, root / "clean-repeat", speculation=False) + require( + compare_campaign(clean, repeat, root / "clean-check")["equal"], + "clean native control is not repeatable", + ) + step = len(plan["forced_tokens"]) - 1 if fault == "missing_observation" else 0 + failed = False + try: + damaged = native( + spec, + plan, + root / "damaged", + speculation=False, + experiment={"fault": fault, "step": step}, + ) + except DiagnosticError: + failed = True + damaged = root / "damaged/run/capture" + applied = private_json(damaged / "fault-applied.json") + require(applied.get("fault") == fault, "native fault was never applied") + if fault in {"kv", "gdn", "conv"}: + require(not failed, "native fault failed before the byte comparator observed it") + report = compare_campaign(clean, damaged, root / "fault-comparison") + require(not report["equal"], "actual GPU corruption escaped the state comparator") + # Detect the intended state component, not a later unrelated output. + comparisons = sorted((root / "fault-comparison").glob("comparison-*.json")) + first = private_json(comparisons[0]) + require( + any( + not row["exact_equal"] and row["boundary"] == applied["component"] + for row in first["components"] + ), + "fault comparison did not identify the corrupted component", + ) + else: + require(failed, "invalid native metadata or missing observation was accepted") + error = private_json(root / "damaged/run/worker-failure.json") + expected = { + "pending": "native committed position/prefix", + "position": "native committed position/prefix", + "version": "invalid native committed state version", + "missing_observation": "backend stopped before the replay finished", + } + require( + error.get("type") == "DiagnosticError" and expected[fault] in error.get("message", ""), + "negative control failed for an unrelated reason", + ) + return ["clean_native_repeat", "actual_fault_applied", "intended_fault_detected"] + + +def chat_identity(root, label, generation="initial"): + return { + "id": digest([str(root), label]), + "generation": digest([str(root), generation]), + "title": "synthetic conformance fixture", + "cwd": str(root), + } + + +def completion_body(spec, chat, prefix, *, stream=False, count=None): + return { + "model": spec["server_config"].get("served_model_name", spec["server_config"]["model"]), + "prompt": prefix, + "max_tokens": count or spec["output_tokens"], + "ignore_eos": True, + "temperature": 0, + "top_k": 1, + "seed": 0, + "return_token_ids": True, + "stream": stream, + "cache_salt": cache_salt(chat), + "kv_transfer_params": { + "qwen_chat": chat, + "qwen_snapshot_abi": spec["binding"]["live_data_abi"], + }, + } + + +def generate( + server, spec, chat, prefix, name, *, count=None, stream=False, cancel_after=None, timeout=600 +): + result = server.client.completion( + "/v1/completions", + completion_body(spec, chat, prefix, stream=stream, count=count), + evidence=server.root / name, + allow_length=True, + cancel_after=cancel_after, + timeout=timeout, + ) + if cancel_after is None: + require( + len(result["token_ids"]) == (count or spec["output_tokens"]), + "native server omitted requested raw token IDs", + ) + return result + + +def same_output(a, b): + require( + a["token_ids"] and a["token_ids"] == b["token_ids"], "native greedy continuation changed" + ) + require(a["finish_reason"] == b["finish_reason"], "native stop boundary changed") + + +def observed(server, *names): + events = read_events(server.root, server.executions) + for name in names: + require( + any(e["event"] == name for e in events), "required native path did not execute: " + name + ) + return events + + +def flush(server, chat): + return request_tail_flush(chat, control_directory=server.root / "control", timeout=120) + + +def capture_request(server, spec, case, prefix, name): + plan = plan_for(spec, case, tokens(spec, 2, 1701)) + require( + plan["prefix"] == prefix, "native capture prompt differs from declared synthetic fixture" + ) + write_private(server.root / (name + ".plan.json"), plan) + write_private(server.root / (name + ".binding.json"), spec["binding"]) + marker = server.root / "capture-request.json" + marker.unlink(missing_ok=True) + write_private( + marker, + { + "name": "state-" + name, + "consumed": len(prefix), + "input_digest": digest(prefix), + "plan": str(server.root / (name + ".plan.json")), + "binding": str(server.root / (name + ".binding.json")), + }, + ) + + +def captured_generation(server, spec, case, chat, prefix, name, *, timeout=600): + capture_request(server, spec, case, prefix, name) + try: + result = generate(server, spec, chat, prefix, name, timeout=timeout) + frame = read_frame(server.root / ("state-" + name)) + require( + frame["pending"] == result["token_ids"][0], + "captured pending token differs from actual published response", + ) + return result + finally: + (server.root / "capture-request.json").unlink(missing_ok=True) + + +def corrupt_restore_recovery(server, spec, case, chat, prefix, baseline, store, root, damaged_key): + # A hung loader previously passed this test when the client's 600-second + # timeout was caught as if it were a deliberate backend rejection. + try: + result = captured_generation( + server, spec, case, chat, prefix, "corrupt-restore", timeout=30 + ) + except BackendResponseError as error: + require( + error.http_status is None or 500 <= error.http_status < 600, + "request/authorization errors do not establish corrupt-cache rejection", + ) + observed(server, "snapshot.load.error") + return [ + "verified_native_snapshot", + "actual_payload_corruption", + "native_restore_rejected_corruption", + ] + # Success is admissible only after detection and correct recomputation, + # durable repair, and a second restart proving that repair is usable. + observed(server, "snapshot.load.error", "state.captured") + same_output(baseline, result) + comparison = compare_frames(server.root / "state-live", server.root / "state-corrupt-restore") + write_private(root / "corrupt-recovery-state.json", comparison) + require(comparison["equal"], "corrupt-cache recomputation changed native logical state") + flush(server, chat) + repaired = store.metadata() + require(damaged_key in repaired.get("head", []), "repair dropped the damaged head block") + require( + repaired.get("publication", {}).get("result") == "committed", + "corrupt-cache replacement was not verified and published", + ) + store.read(damaged_key, repaired["verified_block_size"]) + events = observed(server, "snapshot.publish.return") + require( + any( + e["execution"] == server.executions[-1] and e["event"] == "snapshot.publish.return" + for e in events + ), + "corrupt-cache recovery did not publish a replacement in this server incarnation", + ) + write_private(root / "head-repaired.json", repaired) + # Explicit flush, rather than shutdown's final flush, must have made it durable. + server.restart(crash=True) + reloaded = captured_generation(server, spec, case, chat, prefix, "repair-reload", timeout=30) + same_output(baseline, reloaded) + comparison = compare_frames(server.root / "state-live", server.root / "state-repair-reload") + write_private(root / "repaired-reload-state.json", comparison) + require(comparison["equal"], "repaired snapshot changed native logical state after restart") + events = read_events(server.root, server.executions) + current = [e["event"] for e in events if e["execution"] == server.executions[-1]] + require("snapshot.load.return" in current, "repaired snapshot was not loaded after restart") + require("snapshot.load.error" not in current, "repaired snapshot failed again after restart") + return [ + "verified_native_snapshot", + "actual_payload_corruption", + "native_corruption_detected", + "identical_recomputed_state_and_output", + "verified_repaired_disk_head", + "identical_durable_reload_state_and_output", + ] + + +def eviction_capacity(head, configured_bytes): + """Fit one measured durable head with transfer slack, but not two heads.""" + blocks = len(head.get("head", [])) + size = head.get("verified_block_size", 0) + require(blocks >= 2 and size > 0, "eviction fixture has no measured complete head") + capacity = ((3 * blocks + 1) // 2) * size + require(capacity < configured_bytes, "configured primary cache is too small for this trial") + return capacity + + +def interrupted_write_prefix(spec, case, prefix): + # An ordinary 256-token continuation need not cross a snapshot-block + # boundary. Change a full suffix within the same context budget so a real + # immutable object write is required even at the maximum context length. + count = min(8192, len(prefix)) + changed = prefix[:-count] + tokens(spec, count, case["seed"] + 9419) + require(changed != prefix, "interrupted-write fixture did not change any token") + return changed + + +def pending_tail(status, chat): + """Return measured, unpublished state for this exact chat generation.""" + require( + status.get("schema") == "urn:qwen-r9700:radiance-tail-residency:v1" + and isinstance(status.get("chats"), list), + "invalid native pending-tail observation", + ) + rows = [ + row + for row in status["chats"] + if row.get("chat_id") == chat["id"] and row.get("generation") == chat["generation"] + ] + require(len(rows) <= 1, "duplicate native pending-tail observation") + if not rows: + return None + row = rows[0] + require( + all( + type(row.get(k)) is int and row[k] >= 0 + for k in ("tokens", "durable_tokens", "blocks", "bytes") + ), + "invalid native pending-tail counts", + ) + if row["tokens"] > row["durable_tokens"] and row["blocks"] > 0 and row["bytes"] > 0: + return dict(row) + return None + + +def wait_for_pending_tail(server, chat, *, timeout=10): + deadline = time.monotonic() + timeout + while True: + server.process.check() + try: + status = private_json(server.root / "tail.json") + except FileNotFoundError: + status = None + if status is not None and (row := pending_tail(status, chat)) is not None: + return row + require( + time.monotonic() < deadline, "shutdown fixture has no measured pending snapshot tail" + ) + time.sleep(0.05) + + +def shutdown_seed_prefix(prefix): + # An initial 8K+ response is already due for the normal 8K tail flush. + # Save a predecessor first, so the tested response advances its durable + # head by only 2K tokens and leaves a real RAM-only tail to flush on exit. + require(len(prefix) > 2048, "shutdown fixture requires a durable predecessor") + return prefix[:-2048] + + +def require_shutdown_flush(events, execution): + events = [e for e in events if e["execution"] == execution] + starts = [e for e in events if e["event"] == "snapshot.shutdown.enter"] + ends = [e for e in events if e["event"] == "snapshot.shutdown.return"] + require(len(starts) == len(ends) == 1, "shutdown observation is missing or duplicated") + start, end = starts[0], ends[0] + require( + start["pid"] == end["pid"] and start["monotonic_ns"] < end["monotonic_ns"], + "shutdown observation has invalid process or timing", + ) + during = { + e["event"] + for e in events + if e["pid"] == start["pid"] + and start["monotonic_ns"] < e["monotonic_ns"] < end["monotonic_ns"] + } + require( + {"snapshot.flush.return", "snapshot.publish.return"} <= during, + "pending tail was not flushed and published inside shutdown", + ) + + +def lifecycle(spec, case, root): + variant = case["variant"] + prefix = tokens(spec, case["context"], case["seed"]) + chat = chat_identity(root, "A") + with NativeServer(spec, root / "server", allow_gpu=True) as server: + if variant == "shutdown_pending_tail": + generate(server, spec, chat, shutdown_seed_prefix(prefix), "shutdown-seed") + flush(server, chat) + seed_head = ChatStore(server.root / "data", chat).metadata() + require( + seed_head.get("head"), "shutdown fixture did not establish a durable predecessor" + ) + write_private(root / "head-shutdown-seed.json", seed_head) + baseline = captured_generation(server, spec, case, chat, prefix, "live") + if variant != "shutdown_pending_tail": + flush(server, chat) + store = ChatStore(server.root / "data", chat) + before = store.metadata() + write_private(root / "head-before.json", before) + if variant == "eviction": + original = spec["server_config"]["kv_transfer_config"]["kv_connector_extra_config"][ + "cpu_bytes_to_use" + ] + capacity = eviction_capacity(before, original) + server.stop() + server.variant["primary_cache_bytes"] = capacity + server.start() + warmed = captured_generation(server, spec, case, chat, prefix, "eviction-warm") + same_output(baseline, warmed) + comparison = compare_frames( + server.root / "state-live", server.root / "state-eviction-warm" + ) + write_private(root / "eviction-warm-state.json", comparison) + require(comparison["equal"], "eviction setup changed the baseline native state") + flush(server, chat) + write_private( + root / "eviction-capacity.json", + { + "original_bytes": original, + "trial_bytes": capacity, + "head_blocks": len(before["head"]), + }, + ) + if variant in {"ram", "eviction"}: + for n in range(1 if variant == "ram" else 3): + other = chat_identity(root, f"other-{n}") + other_prefix = prefix if variant == "eviction" else prefix[:8192] + generate(server, spec, other, other_prefix, f"other-{n}") + flush(server, other) + observed(server, "bank.activate.return") + worker = json.loads((server.root / "fair-worker.json").read_text()) + write_private(root / "handover.json", worker) + if variant == "ram": + require( + any(r.get("chat_id") == chat["id"] for r in worker["residency"]["images"]), + "RAM case did not leave the tested chat in RAM", + ) + else: + require( + not any(r.get("chat_id") == chat["id"] for r in worker["residency"]["images"]), + "eviction case failed to evict the tested RAM bank", + ) + elif variant in {"clean_restart", "crash_restart", "shutdown_pending_tail"}: + if variant == "shutdown_pending_tail": + pending = wait_for_pending_tail(server, chat) + write_private(root / "pending-tail-before-shutdown.json", pending) + process_root = server.root / f"process-{server.incarnation - 1}" + server.stop() + stopped = private_json(process_root / "shutdown.json") + require( + not stopped["grace_expired"] and stopped["cleanup_error"] is None, + "clean shutdown exceeded its owned process grace", + ) + events = observed(server, "snapshot.shutdown.return", "snapshot.flush.return") + require_shutdown_flush(events, server.executions[-1]) + durable = store.metadata() + write_private(root / "head-after-shutdown.json", durable) + require( + durable.get("generation") == chat["generation"] + and durable.get("tokens", 0) >= pending["tokens"] + and durable.get("head") + and durable.get("publication", {}).get("result") == "committed", + "shutdown did not publish the measured pending tail", + ) + for key in durable["head"]: + store.read(key, durable["verified_block_size"]) + server.start() + else: + server.restart(crash=variant == "crash_restart") + elif variant in {"corrupt_disk", "missing_disk_block"}: + server.stop() + require(before.get("head"), "disk mutation case has no verified durable head") + path = store.path(before["head"][0]) + preserved = server.root / "original-snapshot-block" + import shutil + + shutil.copyfile(path, preserved) + preserved.chmod(0o600) + if variant == "missing_disk_block": + path.unlink() + else: + with path.open("r+b") as stream: + stream.seek(-1, os.SEEK_END) + old = stream.read(1) + stream.seek(-1, os.SEEK_END) + stream.write(bytes([old[0] ^ 1])) + stream.flush() + os.fsync(stream.fileno()) + write_private( + root / "disk-fault.json", + { + "fault": variant, + "key": path.name, + "original_sha256": hashlib.sha256(preserved.read_bytes()).hexdigest(), + }, + ) + server.start() + if variant == "corrupt_disk": + return corrupt_restore_recovery( + server, spec, case, chat, prefix, baseline, store, root, path.name + ) + elif variant == "interrupted_write": + (server.root / "interrupt-write.arm").write_text("owned qualification only\n") + with ThreadPoolExecutor(max_workers=1) as pool: + future = pool.submit( + generate, + server, + spec, + chat, + interrupted_write_prefix(spec, case, prefix), + "pending-write", + ) + deadline = time.monotonic() + 120 + reached = False + while time.monotonic() < deadline: + if any( + e["event"] == "snapshot.partial_write" + for e in read_events(server.root, server.executions) + ): + reached = True + break + if future.done(): + future.result() + # A tail can remain in RAM until explicitly flushed. + break + time.sleep(0.05) + if not reached: + # Flush in another worker because the real writer is about + # to be held at the injected partial-write boundary. + with ThreadPoolExecutor(max_workers=1) as flush_pool: + flush_future = flush_pool.submit(flush, server, chat) + while time.monotonic() < deadline and not reached: + reached = any( + e["event"] == "snapshot.partial_write" + for e in read_events(server.root, server.executions) + ) + time.sleep(0.05) + server.stop(crash=True) + with contextlib.suppress(RuntimeError, TimeoutError): + flush_future.result(timeout=125) + else: + server.stop(crash=True) + with contextlib.suppress(Exception): + future.result(timeout=10) + require(reached, "interrupted-write fault never reached a real native snapshot write") + (server.root / "interrupt-write.arm").unlink() + # The previous complete head must still be readable and unchanged. + require( + store.metadata().get("head") == before.get("head"), + "interrupted publication replaced the previous head", + ) + server.start() + elif variant == "compaction": + successor = chat_identity(root, "A", "compacted") + new_store = ChatStore(server.root / "data", successor) + new_store.activate() + short_case = {**case, "context": min(8192, case["context"])} + short = tokens(spec, short_case["context"], case["seed"]) + changed = captured_generation(server, spec, short_case, successor, short, "compacted") + flush(server, successor) + require( + not (new_store.generations / chat["generation"]).exists(), + "compaction retained predecessor generation", + ) + fresh = chat_identity(root, "fresh-compacted") + fresh_result = captured_generation( + server, spec, short_case, fresh, short, "fresh-compacted" + ) + same_output(changed, fresh_result) + comparison = compare_frames( + server.root / "state-compacted", server.root / "state-fresh-compacted" + ) + write_private(root / "compaction-state.json", comparison) + require(comparison["equal"], "compaction successor differs from fresh prefill") + observed(server, "snapshot.publish.return", "bank.retire.return") + return [ + "successful_generation_publication", + "old_generation_retired", + "fresh_successor_state_and_output", + ] + elif variant == "cancellation": + interrupted = generate( + server, spec, chat, prefix, "cancel", stream=True, cancel_after=2 + ) + require(interrupted["cancelled"], "cancellation was not exercised") + elif variant != "warm": + raise UnavailableError("unknown cache lifecycle variant") + restore_started = time.monotonic_ns() + restored = captured_generation(server, spec, case, chat, prefix, "restored") + same_output(baseline, restored) + comparison = compare_frames(server.root / "state-live", server.root / "state-restored") + write_private(root / "recovery-state.json", comparison) + require(comparison["equal"], "live/restored native logical state differs") + if variant in { + "clean_restart", + "crash_restart", + "interrupted_write", + "eviction", + "shutdown_pending_tail", + }: + events = observed(server, "snapshot.load.return") + require( + any( + e["event"] == "snapshot.load.return" and e["monotonic_ns"] >= restore_started + for e in events + ), + "restored request did not load a snapshot; an earlier setup read is insufficient", + ) + flush(server, chat) + observed(server, "snapshot.publish.return", "state.captured") + if variant == "missing_disk_block": + require(path.exists(), "missing disk block was not repaired after successful prefill") + repaired = store.metadata() + require( + repaired.get("publication", {}).get("result") == "committed", + "repaired checkpoint was not verified and published", + ) + return [ + "initial_native_state", + "actual_" + variant, + "identical_restored_state", + "identical_greedy_suffix", + "verified_durable_head", + *( + ["measured_pending_tail", "verified_head_before_restart"] + if variant == "shutdown_pending_tail" + else [] + ), + ] + + +def natural(spec, case, root): + prefix = tokens(spec, case["context"], case["seed"]) + chat = chat_identity(root, "natural") + variant = case["variant"] + settings = { + "speculation": ({"speculation": False}, {"speculation": True}), + "verify_head": ({"head": False}, {"head": True, "head_audit": True}), + "graphs": ({"graphs": False}, {"graphs": True}), + "async_experimental": ({"asynchronous": False}, {"asynchronous": True}), + "dynamic_width": ({"dynamic_width": False}, {"dynamic_width": True}), + "head_omission_control": ( + {"head": True, "head_audit": True}, + {"head": True, "head_audit": True, "head_fault": True}, + ), + }[variant] + responses, evidence = [], [] + for index, options in enumerate(settings): + with NativeServer(spec, root / f"server-{index}", allow_gpu=True, **options) as server: + responses.append(generate(server, spec, chat, prefix, "output")) + events = observed(server, "runner.prepare.return", "runner.commit.return") + if options.get("graphs"): + observed(server, "graph.replay.return") + if options.get("speculation"): + require( + any(e["details"].get("draft_tokens", 0) > 0 for e in events), + "no natural speculative verification occurred", + ) + if options.get("head_audit"): + audits = [ + e["details"] for e in events if e["details"].get("head_audit") == "compared" + ] + require(audits, "fast verify head was never compared on actual hidden states") + if options.get("head_fault"): + observed(server, "head.fault_applied") + require( + all(a["omitted_winners"] > 0 and a["different_argmax"] > 0 for a in audits), + "omitted-winner negative control escaped the actual head checker", + ) + evidence.append({"variant": server.variant, "execution": server.executions}) + continue + require( + all( + not a["omitted_winners"] + and not a["different_retained_logits"] + and not a["different_argmax"] + for a in audits + ), + "verify-head discrepancy; actual hidden states/logits preserved", + ) + evidence.append({"variant": server.variant, "execution": server.executions}) + if options.get("dynamic_width"): + widths = { + e["details"].get("draft_tokens") + for e in events + if e["event"] == "runner.prepare.return" + and e["details"].get("draft_tokens", 0) > 0 + } + require( + any(0 < width < 7 for width in widths), + "dynamic verifier never exercised a shorter natural width", + ) + write_private(root / "natural-variants.json", {"variants": evidence}) + same_output(*responses) + return [ + "unforced_greedy_tokens", + "same_prompt_and_seed", + "actual_" + variant, + "exact_output_and_stop_equality", + ] + + +def operator(spec, case, root): + if case["variant"] == "native_dispatch": + if not spec["binding"].get("native_entrypoints"): + raise UnavailableError("no reviewed native export binding for this binary") + from qwen_r9700_lab.conformance_reference import reference_precision + + plan = plan_for(spec, case, tokens(spec, 10, 1701), accepted=[7, 0]) + capture = native(spec, plan, root / "dispatch") + receipt = private_json(capture / "native-receipt.json") + require( + isinstance(receipt.get("native_entrypoints"), str) + and len(receipt["native_entrypoints"]) == 64, + "native export dispatch evidence absent", + ) + report = private_json(capture / "dispatch/dispatch.json") + authenticate(report) + dtype = "fp8" if reference_precision(spec["reference_profile"])["kv_fp8"] else "bf16" + require( + report["sha256"] == receipt["native_entrypoints"], "native dispatch receipt changed" + ) + for symbol in ( + f"attn_prefill_h256_gqa6_{dtype}kv", + f"attn_decode_h256_gqa6_{dtype}kv", + "gdn_chunk_scan_k128_v128_c64_bf16", + ): + require_native_dispatch(report, "r4d." + symbol) + return [ + "exact_reviewed_native_export_inventory", + "native_library_identity", + "real_prefill_decode_GDN_dispatch", + ] + scripts = Path(__file__).resolve().parents[2] / "experiments/radiance-public" + names = { + "mxfp4": ("probe_mxfp4_numerics.py", "mxfp4-probe.json"), + "gdn": ("probe_gdn_numerics.py", "gdn-probe.json"), + "norm_rope": ("probe_norm_rope_numerics.py", "norm-rope-probe.json"), + "attention": ("probe_r4d_attention_numerics.py", "attention-probe.json"), + "sampling": ("probe_dflash_sampling_rng.py", "sampling-probe.json"), + } + script, filename = names[case["variant"]] + data = root / "probe" + data.mkdir(mode=0o700) + write_private(data / "production-profile.json", spec["operator_profile"]) + manifest = {**spec["probe_manifest"], "checkpoint": spec["native_config"]["model"]} + require( + not manifest.get("reference_linear") and not manifest.get("gdn_repair_sha256"), + "campaign probes the deployed repair, not an extra experimental source overlay", + ) + write_private(data / "manifest.json", manifest) + args = [spec["python"], str(scripts / script)] + args += ( + ["--output", str(data / filename)] + if case["variant"] == "sampling" + else ["--root", str(data)] + ) + process = OwnedProcess( + args, + root / "probe-process", + env=worker_environment(spec, root), + timeout=spec["case_timeout_seconds"], + ) + code = process.wait() + report = json.loads((data / filename).read_text()) + require(code == 0, "native operator runner failed; original report/log retained") + validate_operator_report(case["variant"], report) + return [ + "real_native_operator", + "independent_numeric_or_statistical_oracle", + "fixed_predeclared_tolerance", + "nonempty_report_domain", + ] + + +def validate_operator_report(variant, report): + """Check the real probe report, including both sides of RNG fault controls.""" + rows = report.get("rows", report.get("cases")) + require(isinstance(rows, list) and rows, "operator probe produced an empty domain") + if variant in {"mxfp4", "attention"}: + require( + report.get("all_passed") is True + and all( + r.get("finite") is True + and r.get("guard_regions_intact") is True + and 0 <= r["relative_error_vs_fp32"] < 0.02 + for r in rows + ), + "operator numerical or memory guard oracle failed", + ) + elif variant == "gdn": + require( + report.get("all_within_one_percent") is True + and all(r["within_one_percent"] for r in rows), + "native GDN differs from independent recurrence beyond declared diagnostic bound", + ) + elif variant == "norm_rope": + require( + report.get("all_finite") is True + and report["maximum_relative_error"] <= 0.02 + and all(r.get("unrotated_channels_exact", True) for r in rows), + "norm/RoPE oracle failed", + ) + else: + require( + report.get("greedy_exact") is True and report.get("native_target_rows"), + "sampler probe omitted native/greedy rows", + ) + require( + any(r["independent_proposal_noise"] for r in rows), + "independent RNG distribution observations were omitted", + ) + require( + all(r["max_error"] < 0.004 for r in rows if r["independent_proposal_noise"]), + "independent RNG distribution regression", + ) + require( + any(not r["independent_proposal_noise"] for r in rows), + "sampling negative control was omitted", + ) + require( + all(r["max_error"] > 0.01 for r in rows if not r["independent_proposal_noise"]), + "shared-RNG negative control was not detected", + ) + require( + all(r["max_error"] < 0.004 for r in report["native_target_rows"]), + "native target distribution regression", + ) + + +def require_native_dispatch(report, site): + authenticate(report) + require( + any( + row.get("site") == site and row.get("completed") is True and "exception_type" not in row + for row in report.get("calls", []) + ), + "required native export was not observed: " + site, + ) + + +TOOL = { + "type": "function", + "function": { + "name": "record", + "description": "Record the supplied synthetic fixture; no external effects.", + "parameters": { + "type": "object", + "properties": {"number": {"type": "integer"}}, + "required": ["number"], + "additionalProperties": False, + }, + }, +} + + +def tool_body(spec, chat, messages=None, *, thinking=False, stream=True): + return { + "model": spec["server_config"].get("served_model_name", spec["server_config"]["model"]), + "messages": messages + or [ + { + "role": "user", + "content": "Call record with number 37. After its result, reply exactly RECORDED.", + } + ], + "tools": [TOOL], + "tool_choice": "auto", + "temperature": 0, + "top_k": 1, + "seed": 0, + "max_tokens": spec["output_tokens"] * (8 if thinking else 1), + "chat_template_kwargs": {"enable_thinking": thinking}, + "stream": stream, + "cache_salt": cache_salt(chat), + "kv_transfer_params": { + "qwen_chat": chat, + "qwen_snapshot_abi": spec["binding"]["live_data_abi"], + }, + } + + +def record_tool(result): + require( + result["finish_reason"] == "tool_calls" and len(result["tools"]) == 1, + "synthetic model task did not emit the requested tool boundary", + ) + tool = result["tools"][0] + require( + tool["name"] == "record" and tool["parsed_arguments"] == {"number": 37}, + "synthetic tool call has incorrect arguments", + ) + return tool + + +def tool_continuation(body, result): + tool = record_tool(result) + assistant = { + "role": "assistant", + "content": result["content"] or None, + "tool_calls": [ + { + "id": tool["id"], + "type": "function", + "function": {"name": tool["name"], "arguments": tool["arguments"]}, + } + ], + } + if result["reasoning"]: + assistant["reasoning_content"] = result["reasoning"] + return { + **body, + "messages": body["messages"] + + [ + assistant, + {"role": "tool", "tool_call_id": tool["id"], "content": "Recorded 37 successfully."}, + ], + } + + +def fit_tool_context(server, spec, case, body): + """Measure the real chat template, reserving room for the tool/result turn.""" + from tokenizers import Tokenizer + + target = min(case["context"], spec["max_context"] - 3 * spec["output_tokens"] - 512) + messages = [{"role": "system", "content": ""}, *body["messages"]] + request = { + "model": body["model"], + "messages": messages, + "tools": body["tools"], + "chat_template_kwargs": body["chat_template_kwargs"], + "add_special_tokens": False, + "add_generation_prompt": True, + } + base = server.client.json("/tokenize", request) + reserve = len(base["tokens"]) + require(target > reserve + 8, "priority context cannot hold the real tool template") + tokenizer = Tokenizer.from_file(str(Path(spec["native_config"]["model"]) / "tokenizer.json")) + padding = tokens(spec, target - reserve + 8, case["seed"]) + count = target - reserve + for _ in range(8): + messages[0]["content"] = tokenizer.decode(padding[:count], skip_special_tokens=False) + measured = len(server.client.json("/tokenize", request)["tokens"]) + if target - 4 <= measured <= target: + write_private( + server.root / "priority-context.json", + { + "A_prompt_tokens": measured, + "B_prompt_tokens": case["context"], + "A_target": target, + }, + ) + return {**body, "messages": messages} + count = max(1, min(len(padding), count + target - measured)) + raise DiagnosticError("could not fit synthetic priority prompt to the declared window") + + +def protocol(spec, case, root): + variant = case["variant"] + if variant == "parser_fragments": + from qwen_r9700_lab.conformance_parser import qualify_parser + + qualify_parser( + spec["native_config"]["model"], + spec["observer_sources"]["vllm.parser.qwen3"], + root / "parser.json", + ) + qualify_parser( + spec["native_config"]["model"], + spec["observer_sources"]["vllm.parser.qwen3"], + root / "parser-deployed.json", + parser_config={ + "tool_parser_name": spec["server_config"]["tool_call_parser"], + "reasoning_parser_name": spec["server_config"]["reasoning_parser"], + "enable_auto_tools": spec["server_config"]["enable_auto_tool_choice"], + }, + ) + return [ + "actual_Qwen3Parser", + "registered_serving_parser", + "single_token_and_grouped_chunks", + "thinking_tool_boundary", + "incomplete_and_colon_are_not_complete_tools", + ] + if variant == "pi_provider_fragments": + if not spec.get("pi_runtime") or not spec.get("pi_runtime_sha256"): + raise UnavailableError( + "the installed Pi provider artifact has not been bound in this campaign" + ) + write_private( + root / "pi.json", {"provider": spec["pi_runtime"], "sha256": spec["pi_runtime_sha256"]} + ) + driver = Path(__file__).resolve().parents[2] / "tests/conformance_pi_driver.mjs" + process = OwnedProcess( + ["node", str(driver), str(root / "pi.json"), str(root / "pi-result.json")], + root / "pi-process", + env=dict(os.environ), + timeout=120, + ) + code = process.wait() + result = private_json(root / "pi-result.json") + require( + code == 0 and result["passed"] is True and len(result["rows"]) == 5, + "real Pi provider lost or duplicated a fragmented tool event", + ) + return [ + "installed_Pi_provider", + "UTF8_and_SSE_fragments", + "exactly_one_tool_event", + "no_tool_execution", + ] + release_variants = { + "minimal_release": ((True, False), True, True, True), + "minimal_release_full_head": ((True, False), False, True, True), + "repeat_release_full_head": ((False, False), False, True, True), + "minimal_release_target_only": ((True, False), False, False, True), + "repeat_release_target_only": ((False, False), False, False, True), + "minimal_release_target_only_eager": ((True, False), False, False, False), + "repeat_release_target_only_eager": ((False, False), False, False, False), + } + if variant in release_variants: + from qwen_r9700_lab.conformance_protocol import compare_protocol_pair + + observations, head, speculation, graphs = release_variants[variant] + prefix = tokens(spec, case["context"], case["seed"]) + results = [] + for index, observe in enumerate(observations): + with NativeServer( + spec, + root / f"server-{index}", + allow_gpu=True, + observe=observe, + head=head, + speculation=speculation, + graphs=graphs, + ) as server: + results.append( + generate(server, spec, chat_identity(root, "minimal"), prefix, "response") + ) + if observe: + names = ["runner.commit.return"] + if graphs: + names.append("graph.replay.return") + observed(server, *names) + else: + require( + not list(server.root.glob("events-*.jsonl")), + "minimal replay unexpectedly loaded diagnostic observers", + ) + compare_protocol_pair( + *results, root / "release-comparison.json", kind="fresh_release_instances" + ) + return [ + "instrumented_vs_unobserved_release" + if any(observations) + else "two_unobserved_fresh_release_instances", + "fast_head_enabled" if head else "target_verify_head_disabled", + "speculative_decode" if speculation else "target_only", + "graph_replay_requested" if graphs else "eager_execution_requested", + "complete_raw_prompt_and_output_ID_comparison", + "same_greedy_output", + "no_tensor_or_dispatch_hooks_in_minimal", + ] + with NativeServer(spec, root / "server", allow_gpu=True) as server: + chat = chat_identity(root, "protocol") + body = tool_body(spec, chat, thinking=variant == "long_thinking") + body["return_token_ids"] = True + if variant == "long_thinking": + # The preserved 2,048-token attempt ended at its explicit budget. + # Increase this synthetic fixture's budget without changing its + # prompt, sampler, stop rules or treatment of an incomplete answer. + body["max_tokens"] = spec["output_tokens"] * 32 + body["messages"][0]["content"] = ( + "Reason carefully through the sum of squares of integers 1 through 100, " + "checking it with two independent derivations. Then call record with number 37. " + "After its result reply exactly RECORDED." + ) + result = server.client.completion( + "/v1/chat/completions", body, evidence=server.root / "tool" + ) + record_tool(result) + if variant == "long_thinking": + require( + len(result["reasoning"]) >= 100, "long-thinking path was not exercised by the model" + ) + follow = tool_continuation(body, result) + answer = server.client.completion( + "/v1/chat/completions", follow, evidence=server.root / "answer" + ) + require( + answer["finish_reason"] == "stop" and answer["content"].strip() == "RECORDED", + "synthetic tool continuation did not finish correctly", + ) + nonstream = server.client.completion( + "/v1/chat/completions", {**body, "stream": False}, evidence=server.root / "nonstream" + ) + record_tool(nonstream) + from qwen_r9700_lab.conformance_protocol import compare_protocol_pair + + compare_protocol_pair(result, nonstream, server.root / "stream-nonstream-comparison.json") + observed(server, "parser.tool_stream.return", "parser.tool_nonstream.return") + if variant == "long_thinking": + observed(server, "parser.reasoning_stream.return", "parser.reasoning_nonstream.return") + return [ + "actual_native_tool_parser", + "tool_result_continuation", + "stream_nonstream_consistency", + "complete_raw_prompt_and_output_ID_comparison", + "honest_finish_reason", + ] + + +def priority(spec, case, root): + from qwen_r9700_lab.conformance_priority import PriorityLease + + variant = case["variant"] + levels = { + "equal0": (0, 0), + "equal1": (1, 1), + "equal2": (2, 2), + "priority1_owner": (1, 0), + "priority1_waiter": (0, 1), + "priority2_waiter": (0, 2), + }[variant] + a, b = chat_identity(root, "priority-A"), chat_identity(root, "priority-B") + with NativeServer(spec, root / "server", allow_gpu=True) as server: + body = fit_tool_context(server, spec, case, tool_body(spec, a)) + prefix = tokens(spec, case["context"], case["seed"]) + leases = [ + PriorityLease(server.client, chat, spec["binding"]["live_data_abi"], root / name) + for chat, name in ((a, "priority-control-A"), (b, "priority-control-B")) + ] + la, lb = leases + try: + la.start(levels[0]).result() + lb.start(0).result() + with ThreadPoolExecutor(max_workers=2) as pool: + fa = pool.submit( + server.client.completion, + "/v1/chat/completions", + body, + evidence=server.root / "A", + ) + deadline = time.monotonic() + spec["case_timeout_seconds"] + # Wait for actual A generation before B requests ownership. + while time.monotonic() < deadline: + phases = server.root / "fair-phases.json" + if phases.exists(): + status = json.loads(phases.read_text()) + # Actual phase records are inspected below as well; a + # response completing before overlap is a coverage gap. + if '"generate"' in json.dumps(status) and a["id"] in json.dumps(status): + break + require( + not fa.done(), "A finished before overlapping generation could be tested" + ) + time.sleep(0.02) + else: + raise TimeoutError("A never reached observed generation") + + def run_b(): + # Independent Pi processes have independent control queues. + # B's admission must not delay consuming A's tool response. + try: + if levels[1]: + lb.change(levels[1]).result() + return generate(server, spec, b, prefix, "B") + finally: + lb.release().result() + + fb = pool.submit(run_b) + result_a = fa.result(timeout=spec["case_timeout_seconds"]) + a_finished = time.monotonic_ns() + record_tool(result_a) + # Real client-side tool duration: no model output is forced. + if variant == "priority1_owner": + until = time.monotonic() + 3 + while time.monotonic() < until: + require(not fb.done(), "lower-priority chat ran during the retained answer") + time.sleep(0.02) + elif variant.startswith("equal"): + time.sleep(0.25) # explicitly test the short-tool grace + follow = tool_continuation(body, result_a) + a_continuation_submitted = time.monotonic_ns() + continuation = server.client.completion( + "/v1/chat/completions", follow, evidence=server.root / "A-continuation" + ) + a_answer_finished = time.monotonic_ns() + require(continuation["finish_reason"] == "stop", "priority answer failed to finish") + la.release().result() + result_b = fb.result(timeout=spec["case_timeout_seconds"]) + events = observed(server, "scheduler.step.return", "bank.activate.return") + b_steps = [ + e + for e in events + if e["event"] == "scheduler.step.return" + and e["details"].get("active", "").startswith(b["id"] + ":") + and e["details"].get("scheduled") + ] + require(b_steps, "waiting chat never received actual scheduled tokens") + first_b = b_steps[0]["monotonic_ns"] + a_finishes = [ + e["monotonic_ns"] + for e in events + if e["event"] == "response.finish.return" and e["details"].get("chat_id") == a["id"] + ] + require(len(a_finishes) == 2, "native A response completion evidence is incomplete") + write_private( + root / "priority-order.json", + { + "first_B_scheduled_ns": first_b, + "A_tool_finished_ns": a_finished, + "A_continuation_submitted_ns": a_continuation_submitted, + "A_answer_finished_ns": a_answer_finished, + "A_native_response_finished_ns": a_finishes, + "control_failures": [r for lease in leases for r in lease.failures], + }, + ) + # A short intended sleep is insufficient: verify that the actual + # native tool boundary to continuation submission was also short. + if variant.startswith("equal"): + require( + a_continuation_submitted - a_finishes[0] < 2_000_000_000, + "short-tool fixture exceeded the two-second grace; inspect timing receipts", + ) + if variant in {"priority1_waiter", "priority2_waiter"}: + activated = [r for r in lb.confirmations if r["purpose"] == "change"] + require( + len(activated) == 1 and activated[0]["finished_ns"] < a_finishes[0], + "B priority was not confirmed during A generation; overlap was not exercised", + ) + if variant == "priority2_waiter": + require( + first_b < a_finishes[0], + "priority two did not preempt ongoing lower-priority generation", + ) + elif variant == "priority1_owner": + require( + first_b >= a_answer_finished, "priority one owner lost GPU before answer end" + ) + else: + require( + first_b >= a_finishes[0], + "ordinary/priority-one waiter interrupted ongoing generation", + ) + if variant.startswith("equal"): + require(first_b >= a_finishes[-1], "equal priorities lost the short-tool grace") + baseline = generate(server, spec, chat_identity(root, "B-fresh"), prefix, "B-fresh") + same_output(result_b, baseline) + require( + not any(lease.failures for lease in leases), + "priority control had unconfirmed updates; inspect timed HTTP receipts", + ) + finally: + for lease in leases: + lease.close() + return [ + "default_priority_without_control_IPC" if levels == (0, 0) else "real_priority_IPC", + "timed_independent_control_queues", + "overlapping_requests", + "tool_boundary_and_answer_ownership", + "no_resumed_output_corruption", + ] + + +SCENARIOS = { + "native_fault": native_fault, + "forced_reference": forced_reference, + "forced_d7": forced_d7, + "rejected_suffix": rejected_suffix, + "operator": operator, + "natural": natural, + "lifecycle": lifecycle, + "priority": priority, + "protocol": protocol, +} diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_session.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_session.py new file mode 100644 index 0000000..465f91f --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_session.py @@ -0,0 +1,330 @@ +"""Durable fail-closed publication of independently checked private frames. + +The SQLite commit is the authoritative publication point. Worker frame files +remain tentative and no output token is returned before the commit. An external +consumer resumes by revision; exactly-once external side effects require that +consumer to acknowledge/idempotently apply those revisions. +""" + +from __future__ import annotations + +import json +import os +import sqlite3 +import struct +import uuid +from pathlib import Path + +from qwen_r9700_lab.conformance_gate import publication_allowed +from qwen_r9700_lab.conformance_state import ( + archive_frame, + compare_frames, + load_arrays, + private_directory, + read_frame, +) +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + digest, + integer, + remaining_materialized, + seal, + write_private, +) + + +class SessionMismatchError(DiagnosticError): + def __init__(self, receipt: dict): + super().__init__("unchecked transition rejected; no output published") + self.receipt = receipt + + +def sync_directory(path): + descriptor = os.open(path, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + try: + os.fsync(descriptor) + finally: + os.close(descriptor) + + +class CheckedSession: + def __init__(self, root: Path, *, contract: str, required_components: list[str], create=False): + if create: + root.mkdir(mode=0o700) + private_directory(root) + self.root, self.contract, self.required = root, contract, tuple(required_components) + if not self.required or len(set(self.required)) != len(self.required): + raise DiagnosticError("publication gate requires complete unique state coverage") + database = root / "authority.sqlite3" + if create: + fd = os.open(database, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + os.close(fd) + if database.is_symlink() or database.stat().st_mode & 0o077: + raise DiagnosticError("unsafe authority database") + self.db = sqlite3.connect(database, isolation_level=None, timeout=10) + self.db.execute("PRAGMA journal_mode=DELETE") + self.db.execute("PRAGMA synchronous=FULL") + if create: + self.db.executescript(""" + CREATE TABLE metadata (contract TEXT NOT NULL, coverage TEXT NOT NULL); + CREATE TABLE commits ( + revision INTEGER PRIMARY KEY, operation TEXT NOT NULL, + input_digest TEXT NOT NULL, reference_frame TEXT NOT NULL, + candidate_frame TEXT NOT NULL, receipt TEXT NOT NULL, + token_bytes BLOB NOT NULL, stop_reason TEXT); + CREATE TABLE failures (id INTEGER PRIMARY KEY, revision INTEGER NOT NULL, + receipt TEXT NOT NULL); + """) + self.db.execute("INSERT INTO metadata VALUES (?, ?)", (contract, digest(self.required))) + (root / "frames").mkdir(mode=0o700) + sync_directory(root) + sync_directory(root.parent) + row = self.db.execute("SELECT contract,coverage FROM metadata").fetchone() + if row != (contract, digest(self.required)): + self.close() + raise DiagnosticError("authority reference contract changed") + + @property + def revision(self): + return self.db.execute("SELECT COALESCE(MAX(revision),0) FROM commits").fetchone()[0] + + def commit( + self, + reference: Path, + candidate: Path, + *, + base_revision: int, + reference_tokens: tuple[int, ...], + candidate_tokens: tuple[int, ...], + reference_stop: str | None, + candidate_stop: str | None, + ) -> dict: + integer(base_revision) + for tokens in (reference_tokens, candidate_tokens): + if type(tokens) is not tuple: + raise DiagnosticError("tentative output must be immutable") + for token in tokens: + integer(token) + if token >= 2**31: + raise DiagnosticError("invalid output token") + allowed = {None, "eos", "tool_call", "complete"} + if reference_stop not in allowed or candidate_stop not in allowed: + raise DiagnosticError("errors and truncation cannot become successful EOS") + # Workers may reuse their buffers after handing over a transition. The + # authority holds its own verified copies, never a path into a worker. + attempt = self.root / "frames" / uuid.uuid4().hex + attempt.mkdir(mode=0o700) + a = archive_frame(reference, attempt / "reference") + b = archive_frame(candidate, attempt / "candidate") + # Frame files and their own directory entries are synced by the writer. + # Persist both parent links before SQLite can reference this attempt. + sync_directory(attempt) + sync_directory(attempt.parent) + reference, candidate = attempt / "reference", attempt / "candidate" + if any(f["logical"].get("execution_mode") == "forced_token_replay" for f in (a, b)): + raise DiagnosticError("forced diagnostic tokens cannot be published to a session") + required = list(self.required) + if a["coverage"] != required or b["coverage"] != required: + raise DiagnosticError("candidate did not expose the required complete state") + if a["contract"] != self.contract or b["contract"] != self.contract: + raise DiagnosticError("candidate uses a different model contract") + if "sequence.tokens" in required: + _, sequence = load_arrays(reference) + tokens = sequence["sequence.tokens"] + position = sequence.get("sequence.position") + if tokens.dtype.str != "? ORDER BY revision", + (revision,), + ) + ] + + def summary(self): + return { + "revision": self.revision, + "rejected_transitions": self.db.execute("SELECT COUNT(*) FROM failures").fetchone()[0], + "publication": "RUNTIME-CHECKED" if self.revision else "UNPROVED", + "scope": "submitted_frame_and_output_equality", + "native_state_completeness": "UNPROVED", + "formal_backend_equivalence": "UNPROVED", + } + + def close(self): + self.db.close() + + +def compare_campaign(reference: Path, candidate: Path, output: Path) -> dict: + """Compare a sealed schedule, including initial prefill, in causal order.""" + from qwen_r9700_lab.diagnostic_contract import authenticate, private_json + + schedules = [private_json(p / "schedule.json") for p in (reference, candidate)] + for schedule in schedules: + authenticate(schedule) + if schedule.get("schema") != "urn:qwen:conformance-schedule:v1" or not schedule.get( + "frames" + ): + raise DiagnosticError("missing or empty execution schedule") + from qwen_r9700_lab.conformance_replay import scheduled_inputs, validate_plan + + plans = [validate_plan(private_json(p / "plan.json")) for p in (reference, candidate)] + if plans[0] != plans[1] or schedules[0]["contract"] != schedules[1]["contract"]: + raise DiagnosticError("reference and candidate executed different schedules") + expected = list(scheduled_inputs(plans[0])) + for root, schedule in zip((reference, candidate), schedules, strict=True): + if schedule.get("plan") != plans[0]["sha256"] or len(schedule["frames"]) != len(expected): + raise DiagnosticError("schedule omitted required observations") + if ( + schedule.get("initial_state") != "independent_zero_state" + or schedule.get("published_to_session") is not False + ): + raise DiagnosticError("replay did not declare independent initial prefill") + for actual, required in zip(schedule["frames"], expected, strict=True): + if {k: v for k, v in actual.items() if k != "sha256"} != required: + raise DiagnosticError("schedule diverged from its declared plan") + frame = read_frame(root / required["name"]) + if frame["sha256"] != actual["sha256"] or frame["coverage"] != schedule["coverage"]: + raise DiagnosticError("schedule frame changed or is incomplete") + if schedules[0]["coverage"] != schedules[1]["coverage"]: + raise DiagnosticError("reference and candidate coverage differs") + output.mkdir(mode=0o700) + rows, first = [], None + for index, entry in enumerate(expected): + name = entry["name"] + result = compare_frames(reference / name, candidate / name) + write_private(output / f"comparison-{index:06d}.json", result) + rows.append({"index": index, "equal": result["equal"], "sha256": result["sha256"]}) + if first is None and not result["equal"]: + first = {"frame": index, **result["first_difference"]} + report = seal( + { + "schema": "urn:qwen:conformance-campaign-result:v1", + "frames": rows, + "first_difference": first, + "equal": first is None, + "formal_backend_equivalence": "UNPROVED", + } + ) + write_private(output / "report.json", report) + return report diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_shm.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_shm.py new file mode 100644 index 0000000..ef219f1 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_shm.py @@ -0,0 +1,68 @@ +"""Own one test engine's named offload mapping across process teardown. + +The runtime can leave its mmap name behind after an abort/crash. Unlink only +the fresh, explicitly claimed conformance engine name, after its owned process +has been stopped. Unlinking does not change any still-open mapping: the OS +reclaims those pages when the last reference closes. Never scan or clear other +engines' shared memory, and never apply this to production engine identities. +""" + +from __future__ import annotations + +import os +import re +import stat +from pathlib import Path + +from qwen_r9700_lab.diagnostic_contract import DiagnosticError + + +class OwnedOffloadRegion: + def __init__(self, engine_id: str, *, directory=Path("/dev/shm")): + if not isinstance(engine_id, str) or not re.fullmatch( + r"conformance-[0-9a-f]{64}", engine_id + ): + raise DiagnosticError( + "offload cleanup requires an isolated conformance engine identity" + ) + self.engine_id = engine_id + self.name = f"vllm_offload_{engine_id}.mmap" + self.path = Path(directory) / self.name + self.fd = os.open(directory, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW | os.O_CLOEXEC) + try: + try: + os.stat(self.name, dir_fd=self.fd, follow_symlinks=False) + except FileNotFoundError: + pass + else: + raise DiagnosticError("qualification offload name already exists; not owned") + except BaseException: + os.close(self.fd) + self.fd = None + raise + + def release(self): + """Release the name after process close; report unlink, not proven page reclamation.""" + if self.fd is None: + raise DiagnosticError("offload ownership was already released") + try: + try: + info = os.stat(self.name, dir_fd=self.fd, follow_symlinks=False) + except FileNotFoundError: + return {"status": "absent", "engine_id": self.engine_id, "path": str(self.path)} + if not stat.S_ISREG(info.st_mode) or info.st_uid != os.getuid() or info.st_nlink != 1: + raise DiagnosticError("qualification offload file type or ownership changed") + os.unlink(self.name, dir_fd=self.fd) + return { + "status": "unlinked", + "engine_id": self.engine_id, + "path": str(self.path), + "device": info.st_dev, + "inode": info.st_ino, + "bytes": info.st_size, + "allocated_bytes_before_unlink": info.st_blocks * 512, + "page_reclamation": "when the last mapping/file reference closes", + } + finally: + os.close(self.fd) + self.fd = None diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_stage_isolation.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_stage_isolation.py new file mode 100644 index 0000000..da434b0 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_stage_isolation.py @@ -0,0 +1,200 @@ +"""Check one numerical substitution from a common, immutable reference cut. + +Native adapters must supply the captured reference inputs/state and implement +the stage and reference remainder. This checker grants no native qualification +merely because an adapter exists or a whole-model replay passed. +""" + +from dataclasses import dataclass +from typing import Protocol + +import numpy as np + +from qwen_r9700_lab.conformance_topk import compare_rows, summarize_logits +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, authenticate, seal + + +def validate_capture_bridge(reference, candidate, original_fixture, capture_fixture, capture): + """Bind an instrumented prefix of a replay to the observed release outputs. + + This checks outputs and coverage, not hidden-state equality or an operator's + isolated correctness. Those are separate obligations. + """ + for record in (reference, candidate, original_fixture, capture_fixture, capture): + authenticate(record) + count = len(capture_fixture["output"]) - 1 + if ( + count < 1 + or capture_fixture["prefix"] != original_fixture["prefix"] + or capture_fixture["output"] != original_fixture["output"][: count + 1] + or reference["continuation"] != original_fixture["sha256"] + or candidate["continuation"] != capture_fixture["sha256"] + ): + raise DiagnosticError("capture and release do not consume the same saved prefix") + if len(candidate["rows"]) != count or len(reference["rows"]) < count: + raise DiagnosticError("capture bridge has an incomplete position domain") + start = len(capture_fixture["prefix"]) + expected = list(range(start, start + count)) + observed = [r["absolute_position"] for r in candidate["rows"]] + captured = [p for b in capture["batches"] for p in b["positions"]] + if observed != expected or captured != expected or capture["positions"] != count: + raise DiagnosticError("capture bridge position coverage changed") + if not any(k.startswith("inductor/") and v > 0 for k, v in capture["counts"].items()): + raise DiagnosticError("no actual compiled launch was captured") + if candidate["prefill"] != reference["prefill"]: + raise DiagnosticError("instrumented prefill differs from the release") + if candidate["rows"] != reference["rows"][:count]: + raise DiagnosticError("instrumented decode differs from the release") + for left, right in zip(reference["rows"][:count], candidate["rows"], strict=True): + comparison = compare_rows(left["logits"], right["logits"]) + if not comparison["full_logits_exact"]: + raise DiagnosticError("captured full-vocabulary digest differs") + return seal( + { + "schema": "qwen.compiled-capture-output-bridge.v1", + "status": "MATCHED", + "positions": count, + "full_logits_exact": count, + "prefill_exact": True, + "same_saved_inputs": True, + "release_rows_sha256": reference["sha256"], + "captured_rows_sha256": candidate["sha256"], + "capture_manifest_sha256": capture["sha256"], + "reference_fixture_sha256": original_fixture["sha256"], + "capture_fixture_sha256": capture_fixture["sha256"], + "scope": ( + "Observed compiled outputs and position coverage only; " + "not isolated-stage or hidden-state qualification" + ), + } + ) + + +@dataclass +class Cut: + position: int + layer: int | None + stage: str + inputs: dict[str, np.ndarray] + state: dict[str, np.ndarray] + reference_output: dict[str, np.ndarray] + reference_next_state: dict[str, np.ndarray] + reference_logits: np.ndarray + + +class Adapter(Protocol): + def stage(self, arm, inputs, state): + """Return (output arrays, next state arrays), without modifying inputs.""" + + def remainder(self, output, next_state): + """Evaluate the pinned reference remainder from an isolated state copy.""" + + +def clone(arrays): + return {k: np.array(v, copy=True, order="K") for k, v in arrays.items()} + + +def exact(a, b): + if a.keys() != b.keys(): + return False + return all( + a[k].dtype == b[k].dtype + and a[k].shape == b[k].shape + and a[k].tobytes(order="C") == b[k].tobytes(order="C") + for k in a + ) + + +def logits_exact(a, b): + return exact({"logits": a}, {"logits": b}) + + +def evaluate(cut: Cut, adapter: Adapter, *, arms=("old", "fixed")): + if not cut.inputs or not cut.reference_output: + raise DiagnosticError("an isolated comparison needs observed inputs and outputs") + logits = cut.reference_logits + if ( + logits.ndim != 1 + or logits.size < 21 + or logits.dtype != np.float32 + or not np.isfinite(logits).all() + ): + raise DiagnosticError("the isolated comparison requires full finite vocabulary logits") + # This is a mandatory check of the remainder, not an assumption that any + # function called 'reference' implements the reference calculation. + replayed = adapter.remainder(clone(cut.reference_output), clone(cut.reference_next_state)) + if not logits_exact(logits, replayed): + raise DiagnosticError( + "reference remainder does not reproduce the admitted reference logits" + ) + reference_summary = summarize_logits(logits) + results = {} + for arm in arms: + inputs, initial = clone(cut.inputs), clone(cut.state) + output, state = adapter.stage(arm, inputs, initial) + if not exact(inputs, cut.inputs): + raise DiagnosticError("candidate modified an input outside the declared state") + # State changes are never hidden behind an output-only comparison. + output_equal = exact(output, cut.reference_output) + state_equal = exact(state, cut.reference_next_state) + if output_equal and state_equal: + observed = logits + propagation = "identical output and state imply identical pinned reference remainder" + else: + observed = adapter.remainder(clone(output), clone(state)) + propagation = "reference remainder evaluated on this isolated substitution" + if observed.shape != logits.shape or not np.isfinite(observed).all(): + raise DiagnosticError( + "candidate remainder did not produce full finite vocabulary logits" + ) + results[arm] = { + "position": cut.position, + "layer": cut.layer, + "stage": cut.stage, + "input_exact": True, + "reference_remainder_verified": True, + "stage_output_exact": output_equal, + "stage_state_exact": state_equal, + "full_logits_exact": logits_exact(logits, observed), + "topk": compare_rows(reference_summary, summarize_logits(observed)), + "propagation": propagation, + } + return results + + +def aggregate_stage(records, positions, layers): + """Count a position only if every declared layer instance meets the metric.""" + positions, layers = tuple(positions), tuple(layers) + if len(positions) != 320 or len(set(positions)) != 320 or not layers: + raise DiagnosticError("isolated stage publication requires 320 unique positions") + if len(set(layers)) != len(layers): + raise DiagnosticError("duplicate declared layer instance") + expected = {(p, layer) for p in positions for layer in layers} + actual = {(r["position"], r["layer"]) for r in records} + if actual != expected or len(records) != len(expected): + raise DiagnosticError( + "isolated stage coverage is missing, duplicated or outside its domain" + ) + if not all(r["input_exact"] and r["reference_remainder_verified"] for r in records): + raise DiagnosticError("the stage comparisons do not share verified reference inputs") + by_position = {p: [r for r in records if r["position"] == p] for p in positions} + return { + "positions": 320, + "layer_instances": list(layers), + "evaluations": len(records), + "isolated_inputs_verified": True, + "reference_remainder_verified": True, + "criterion": "a position passes only when every declared layer instance passes", + "stage_output_exact": sum( + all(r["stage_output_exact"] for r in rs) for rs in by_position.values() + ), + "stage_state_exact": sum( + all(r["stage_state_exact"] for r in rs) for rs in by_position.values() + ), + "top20_set_exact": sum( + all(r["topk"]["20"]["set_exact"] for r in rs) for rs in by_position.values() + ), + "top20_order_exact": sum( + all(r["topk"]["20"]["ranked_exact"] for r in rs) for rs in by_position.values() + ), + } diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_state.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_state.py new file mode 100644 index 0000000..ae3e1b5 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_state.py @@ -0,0 +1,440 @@ +"""Canonical, private tensor frames and streaming exact comparison (CPU only). + +Frames describe logical values; physical block numbers never enter equality. +Raw bytes are compared even when their digests match. Numerical distances are +diagnostics, never an excuse for passing an unequal exact comparison. +""" + +from __future__ import annotations + +import fcntl +import hashlib +import math +import os +import stat +from collections.abc import Iterable, Mapping +from pathlib import Path + +import numpy as np + +from qwen_r9700_lab.diagnostic_contract import ( + DiagnosticError, + authenticate, + digest, + integer, + private_json, + require_name, + require_sha, + seal, + write_private, +) +from qwen_r9700_lab.exact_fp8_metrics import DTYPES as FP8_METRIC_DTYPES +from qwen_r9700_lab.exact_fp8_metrics import MAX_ELEMENTS as FP8_METRIC_MAX +from qwen_r9700_lab.exact_fp8_metrics import chunk_metrics as fp8_chunk_metrics + +SCHEMA = "urn:qwen:canonical-state-frame:v1" +DTYPES = { + name: np.dtype(name) + for name in (" None: + info = path.lstat() + if not stat.S_ISDIR(info.st_mode) or info.st_uid != os.getuid() or info.st_mode & 0o077: + raise DiagnosticError("evidence directory must be private and owned") + + +def open_blob(root: Path, name: str): + if len(name) != 68 or not name.endswith(".bin"): + raise DiagnosticError("invalid canonical blob name") + require_sha(name[:-4]) + fd = os.open(root / name, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) + info = os.fstat(fd) + if not stat.S_ISREG(info.st_mode) or info.st_uid != os.getuid() or info.st_mode & 0o077: + os.close(fd) + raise DiagnosticError("tensor evidence must be private and owned") + return os.fdopen(fd, "rb") + + +def values(raw: bytes, dtype: str) -> np.ndarray: + result = np.frombuffer(raw, dtype=DTYPES[dtype]) + if dtype == "bf16": + return (result.astype("> 3) & 15, code & 7 + bias = 8 if dtype.endswith("fnuz") else 7 + decoded = np.ldexp(1.0 + mantissa / 8, exponent.astype(int) - bias) + decoded = np.where(exponent == 0, np.ldexp(mantissa / 8, 1 - bias), decoded) + decoded = np.where(code & 128, -decoded, decoded) + nan = code == 128 if bias == 8 else (exponent == 15) & (mantissa == 7) + return np.where(nan, np.nan, decoded) + return result.astype(np.float64) + + +class FrameWriter: + """A frame becomes readable only after its final manifest is published.""" + + def __init__( + self, + root: Path, + *, + contract: str, + execution: str, + adapter: str, + input_digest: str, + phase: str, + consumed: int, + pending: int | None, + expected: Iterable[str], + logical: Mapping | None = None, + ): + for item in (contract, execution, adapter, input_digest): + require_sha(item) + if phase not in PHASES: + raise DiagnosticError("unsupported capture phase") + integer(consumed) + if pending is not None: + integer(pending) + names = tuple(require_name(v) for v in expected) + if not names or len(set(names)) != len(names): + raise DiagnosticError("frame coverage must be nonempty and unique") + root.mkdir(mode=0o700) + private_directory(root) + self.root, self.expected, self.components = root, names, {} + self.header = { + "schema": SCHEMA, + "contract": contract, + "execution": execution, + "adapter": adapter, + "input_digest": input_digest, + "phase": phase, + "consumed": consumed, + "pending": pending, + "logical": dict(logical or {}), + } + self.finished = False + + def add(self, name: str, raw: bytes, *, dtype: str, shape: Iterable[int]) -> None: + if type(raw) is not bytes: + raise DiagnosticError("unsupported tensor representation") + self.add_stream(name, (raw,), dtype=dtype, shape=shape) + + def add_stream(self, name: str, chunks, *, dtype: str, shape: Iterable[int]) -> None: + """Write bounded chunks; incomplete/extra data cannot publish a frame.""" + if self.finished or name not in self.expected or name in self.components: + raise DiagnosticError("duplicate, unexpected or late tensor observation") + shape = list(shape) + for dimension in shape: + integer(dimension) + if dtype not in DTYPES: + raise DiagnosticError("unsupported tensor representation") + expected = math.prod(shape) * DTYPES[dtype].itemsize + name_digest = digest(name) + filename = name_digest + ".bin" + fd = os.open(self.root / filename, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + size, checksum = 0, hashlib.sha256() + with os.fdopen(fd, "wb") as f: + for chunk in chunks: + if not isinstance(chunk, (bytes, memoryview)): + raise DiagnosticError("unsupported tensor chunk") + size += len(chunk) + if size > expected: + raise DiagnosticError("tensor shape does not describe its payload") + f.write(chunk) + checksum.update(chunk) + if size != expected: + raise DiagnosticError("tensor shape does not describe its payload") + f.flush() + os.fsync(f.fileno()) + self.components[name] = { + "file": filename, + "dtype": dtype, + "shape": shape, + "nbytes": size, + "sha256": checksum.hexdigest(), + } + + def array(self, name: str, value: np.ndarray) -> None: + array = np.asarray(value) + dtype = array.dtype.newbyteorder("<") + if dtype.str not in DTYPES: + raise DiagnosticError("unsupported canonical array dtype") + contiguous = np.ascontiguousarray(array, dtype=dtype) + raw = memoryview(contiguous.reshape(-1)).cast("B") + self.add_stream( + name, + (raw[start : start + 1024 * 1024] for start in range(0, len(raw), 1024 * 1024)), + dtype=dtype.str, + shape=array.shape, + ) + + def finish(self) -> dict: + if self.finished or set(self.components) != set(self.expected): + raise DiagnosticError("cannot publish incomplete or already published frame") + document = seal( + {**self.header, "coverage": list(self.expected), "components": self.components} + ) + write_private(self.root / "frame.json", document) + fd = os.open(self.root, os.O_RDONLY | os.O_DIRECTORY) + try: + os.fsync(fd) + finally: + os.close(fd) + self.finished = True + return document + + +def read_frame(root: Path) -> dict: + private_directory(root) + frame = private_json(root / "frame.json") + authenticate(frame) + if set(frame) != { + "schema", + "contract", + "execution", + "adapter", + "input_digest", + "phase", + "consumed", + "pending", + "logical", + "coverage", + "components", + "sha256", + }: + raise DiagnosticError("frame has incomplete metadata") + if frame["schema"] != SCHEMA or frame["phase"] not in PHASES: + raise DiagnosticError("unsupported canonical frame") + for name in ("contract", "execution", "adapter", "input_digest"): + require_sha(frame[name]) + integer(frame["consumed"]) + if not isinstance(frame["logical"], dict) or not isinstance(frame["components"], dict): + raise DiagnosticError("invalid logical state metadata") + if frame["pending"] is not None: + integer(frame["pending"]) + coverage = frame["coverage"] + if ( + not isinstance(coverage, list) + or not coverage + or any(not isinstance(v, str) for v in coverage) + ): + raise DiagnosticError("invalid frame coverage") + if len(set(coverage)) != len(coverage) or set(coverage) != set(frame["components"]): + raise DiagnosticError("incomplete frame coverage") + for name, descriptor in frame["components"].items(): + require_name(name) + if set(descriptor) != {"file", "dtype", "shape", "nbytes", "sha256"}: + raise DiagnosticError("incomplete tensor descriptor") + require_sha(descriptor["sha256"]) + if descriptor["file"] != digest(name) + ".bin" or descriptor["dtype"] not in DTYPES: + raise DiagnosticError("tensor name or dtype mismatch") + if not isinstance(descriptor["shape"], list): + raise DiagnosticError("invalid tensor shape") + for dimension in descriptor["shape"]: + integer(dimension) + if ( + integer(descriptor["nbytes"]) + != math.prod(descriptor["shape"]) * DTYPES[descriptor["dtype"]].itemsize + ): + raise DiagnosticError("tensor size mismatch") + return frame + + +def compare_frames(left: Path, right: Path, *, chunk_bytes: int = 1024 * 1024) -> dict: + """Bounded-memory exact comparison, including numerics and first byte offset.""" + a, b = read_frame(left), read_frame(right) + for key in ("contract", "input_digest", "phase", "coverage"): + if a[key] != b[key]: + raise DiagnosticError("different inputs, semantics or observation coverage") + if chunk_bytes < 8 or chunk_bytes % 8: + raise DiagnosticError("comparison chunks must align every admitted dtype") + results, first = [], None + metadata_equal = all(a[k] == b[k] for k in ("consumed", "pending", "logical")) + if not metadata_equal: + first = {"boundary": "logical_state", "kind": "metadata"} + for name in a["coverage"]: + da, db = a["components"][name], b["components"][name] + representation_equal = all(da[k] == db[k] for k in ("dtype", "shape", "nbytes")) + if not representation_equal: + raise DiagnosticError("tensor representations require an explicit adapter") + hashes = [hashlib.sha256(), hashlib.sha256()] + byte_offset, first_offset, mismatch_count, nonfinite = 0, None, 0, [0, 0] + max_abs, squared, ref_squared, count = ( + np.longdouble(0), + np.longdouble(0), + np.longdouble(0), + 0, + ) + with open_blob(left, da["file"]) as fa, open_blob(right, db["file"]) as fb: + identities = [os.fstat(f.fileno()) for f in (fa, fb)] + while True: + ra, rb = fa.read(chunk_bytes), fb.read(chunk_bytes) + if not ra and not rb: + break + if len(ra) != len(rb) or len(ra) % DTYPES[da["dtype"]].itemsize: + raise DiagnosticError("tensor payload truncated") + hashes[0].update(ra) + hashes[1].update(rb) + different = np.frombuffer(ra, dtype=np.uint8) != np.frombuffer(rb, dtype=np.uint8) + if different.any(): + if first_offset is None: + first_offset = byte_offset + int(np.flatnonzero(different)[0]) + mismatch_count += int(np.count_nonzero(different)) + if ( + da["dtype"] in FP8_METRIC_DTYPES + and len(ra) <= FP8_METRIC_MAX + and np.finfo(np.longdouble).nmant >= 63 + ): + metrics = fp8_chunk_metrics(ra, rb, da["dtype"]) + nonfinite[0] += metrics["nonfinite"][0] + nonfinite[1] += metrics["nonfinite"][1] + max_abs = max(max_abs, metrics["max_abs"]) + squared += metrics["squared"] + ref_squared += metrics["reference_squared"] + count += metrics["count"] + elif da["dtype"] != "bytes": + va, vb = values(ra, da["dtype"]), values(rb, db["dtype"]) + finite = np.isfinite(va) & np.isfinite(vb) + nonfinite[0] += int(np.count_nonzero(~np.isfinite(va))) + nonfinite[1] += int(np.count_nonzero(~np.isfinite(vb))) + delta = va[finite].astype(np.longdouble) - vb[finite].astype(np.longdouble) + if delta.size: + max_abs = max(max_abs, np.max(np.abs(delta))) + squared += np.sum(delta * delta) + ref_squared += np.sum(va[finite].astype(np.longdouble) ** 2) + count += delta.size + byte_offset += len(ra) + for f, before in zip((fa, fb), identities, strict=True): + after = os.fstat(f.fileno()) + if (before.st_size, before.st_mtime_ns, before.st_ctime_ns) != ( + after.st_size, + after.st_mtime_ns, + after.st_ctime_ns, + ): + raise DiagnosticError("tensor changed during comparison") + if byte_offset != da["nbytes"] or any( + h.hexdigest() != d["sha256"] for h, d in zip(hashes, (da, db), strict=True) + ): + raise DiagnosticError("tensor content does not match its sealed descriptor") + equal = mismatch_count == 0 and nonfinite == [0, 0] + row = { + "boundary": name, + "exact_equal": equal, + "differing_bytes": mismatch_count, + "first_byte_offset": first_offset, + "nonfinite": nonfinite, + "max_abs": finite_metric(max_abs), + "rmse": finite_metric(np.sqrt(squared / count)) if count else None, + "relative_l2": finite_metric(np.sqrt(squared / ref_squared)) if ref_squared else None, + } + results.append(row) + if not equal and first is None: + first = {"boundary": name, "kind": "tensor", "byte_offset": first_offset} + return seal( + { + "schema": "urn:qwen:canonical-state-comparison:v1", + "equal": first is None, + "first_difference": first, + "components": results, + "metadata_equal": metadata_equal, + "reference_frame": a["sha256"], + "candidate_frame": b["sha256"], + "scope": "observed_exact_logical_state", + "formal_backend_equivalence": "UNPROVED", + } + ) + + +def finite_metric(value): + """Overflowing diagnostics never turn an exact mismatch into a pass.""" + return float(value) if abs(value) <= np.finfo(np.float64).max else "overflow" + + +def load_arrays(root: Path) -> tuple[dict, dict[str, np.ndarray]]: + frame = read_frame(root) + arrays = {} + for name, d in frame["components"].items(): + with open_blob(root, d["file"]) as f: + before = os.fstat(f.fileno()) + raw = f.read() + after = os.fstat(f.fileno()) + if (before.st_size, before.st_mtime_ns, before.st_ctime_ns) != ( + after.st_size, + after.st_mtime_ns, + after.st_ctime_ns, + ): + raise DiagnosticError("frame payload changed while reading") + if len(raw) != d["nbytes"] or hashlib.sha256(raw).hexdigest() != d["sha256"]: + raise DiagnosticError("frame payload is corrupt") + arrays[name] = np.frombuffer(raw, dtype=DTYPES[d["dtype"]]).reshape(d["shape"]).copy() + return frame, arrays + + +def archive_frame(source: Path, destination: Path, *, reflink: bool = False) -> dict: + """Create an independent, durable copy and authenticate its actual payload. + + The optional Linux FICLONE path still reads and hashes every destination + byte. It requires filesystem support: there is no silent full-copy fallback + that could exhaust a caller's storage budget. Neither path uses hard links. + """ + f = read_frame(source) + writer = FrameWriter( + destination, + **{ + k: f[k] + for k in ( + "contract", + "execution", + "adapter", + "input_digest", + "phase", + "consumed", + "pending", + "logical", + ) + }, + expected=f["coverage"], + ) + # Stream instead of allocating a complete 250K frame in RAM. + for name in f["coverage"]: + d = f["components"][name] + h, size = hashlib.sha256(), 0 + with open_blob(source, d["file"]) as src: + before = os.fstat(src.fileno()) + fd = os.open(destination / d["file"], os.O_RDWR | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(fd, "w+b") as dst: + if reflink: + # Linux fs.h: FICLONE = _IOW(0x94, 9, int). + fcntl.ioctl(dst.fileno(), 0x40049409, src.fileno()) + # Verify the clone itself, not just the source or manifest. + while chunk := dst.read(1024 * 1024): + h.update(chunk) + size += len(chunk) + else: + while chunk := src.read(1024 * 1024): + h.update(chunk) + size += len(chunk) + dst.write(chunk) + dst.flush() + os.fsync(dst.fileno()) + after = os.fstat(src.fileno()) + if (before.st_size, before.st_mtime_ns, before.st_ctime_ns) != ( + after.st_size, + after.st_mtime_ns, + after.st_ctime_ns, + ): + raise DiagnosticError("frame changed during archival") + if h.hexdigest() != d["sha256"] or size != d["nbytes"]: + raise DiagnosticError("frame payload changed before archival") + writer.components[name] = dict(d) + result = writer.finish() + if result != f: + raise DiagnosticError("archival changed canonical frame metadata") + return result diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_topk.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_topk.py new file mode 100644 index 0000000..4eef538 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_topk.py @@ -0,0 +1,294 @@ +"""Aligned M1/M8 logit comparisons; no inference or token decoding at import. + +Ordering is score descending, token ID ascending. Exact membership, ordering, +overlap and tied cutoffs are different measurements and are reported separately. +This is observed equivalence, never a proof for unobserved model inputs. +""" + +from __future__ import annotations + +import hashlib + +import numpy as np + +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, authenticate, digest, seal + +KS = (1, 10, 20) + + +def require(condition, message): + if not condition: + raise DiagnosticError(message) + + +def summarize_logits(logits): + """Keep exact leading scores and every boundary tie, without serializing text.""" + values = np.asarray(logits) + require(values.ndim == 1 and values.size >= 21, "incomplete vocabulary row") + require(values.dtype == np.float32, "logits must preserve the native FP32 output") + require(np.all(np.isfinite(values)), "non-finite full-head logits") + threshold = np.partition(values, values.size - 21)[-21] + ids = np.flatnonzero(values >= threshold) + order = np.lexsort((ids, -values[ids])) + ids = ids[order] + scores = values[ids] + return { + "vocabulary": int(values.size), + "logits_sha256": hashlib.sha256(values.astype("= 21, "incomplete top-k evidence") + require(len(set(ids)) == len(ids), "duplicate vocabulary IDs") + require(all(type(i) is int and 0 <= i < row["vocabulary"] for i in ids), "invalid ID") + require(all(np.isfinite(x) for x in scores), "non-finite retained score") + require( + list(zip(ids, scores, strict=True)) + == sorted(zip(ids, scores, strict=True), key=lambda x: (-x[1], x[0])), + "noncanonical ranking", + ) + for k in KS: + cutoff = scores[k - 1] + require( + row["boundary_ties"][str(k)] == sum(x == cutoff for x in scores), + "missing tied boundary candidates", + ) + + +def compare_rows(reference, candidate): + validate_row(reference) + validate_row(candidate) + require(reference["vocabulary"] == candidate["vocabulary"], "vocabulary changed") + result = {"full_logits_exact": reference["logits_sha256"] == candidate["logits_sha256"]} + for k in KS: + left, right = reference["ids"][:k], candidate["ids"][:k] + left_ties = { + i + for i, score in zip(reference["ids"], reference["scores"], strict=True) + if score >= reference["scores"][k - 1] + } + right_ties = { + i + for i, score in zip(candidate["ids"], candidate["scores"], strict=True) + if score >= candidate["scores"][k - 1] + } + result[str(k)] = { + "set_exact": set(left) == set(right), + "ranked_exact": left == right, + "overlap": len(set(left) & set(right)), + "inclusive_tie_set_exact": left_ties == right_ties, + "reference_boundary_tied": reference["boundary_ties"][str(k)] > 1, + "candidate_boundary_tied": candidate["boundary_ties"][str(k)] > 1, + "retained_scores_exact": left == right + and reference["scores"][:k] == candidate["scores"][:k], + } + return result + + +def aggregate(comparisons): + require(bool(comparisons), "empty equivalence measurement") + total = len(comparisons) + result = { + "positions": total, + "full_logits_exact": sum(r["full_logits_exact"] for r in comparisons), + } + for k in KS: + rows = [r[str(k)] for r in comparisons] + result[str(k)] = { + **{ + key: sum(row[key] for row in rows) + for key in ( + "set_exact", + "ranked_exact", + "inclusive_tie_set_exact", + "reference_boundary_tied", + "candidate_boundary_tied", + "retained_scores_exact", + ) + }, + "set_exact_percent": 100 * sum(row["set_exact"] for row in rows) / total, + "ranked_exact_percent": 100 * sum(row["ranked_exact"] for row in rows) / total, + "mean_overlap_tokens": sum(row["overlap"] for row in rows) / total, + "mean_overlap_percent": 100 * sum(row["overlap"] for row in rows) / (total * k), + } + return result + + +def compare_saved_rows(before, after, *, target_rows): + """Compare one arm across revisions without assuming its reference is unchanged.""" + for report in (before, after): + authenticate(report) + require( + report.get("schema") == "urn:qwen:d7-equivalence-private-rows:v1", + "wrong saved replay evidence", + ) + require(type(target_rows) is int and target_rows in (1, 8), "unsupported target row count") + require(before["continuation"] == after["continuation"], "saved replay histories differ") + require( + len(before["rows"]) == len(after["rows"]) > 0, + "saved replay lengths differ or are empty", + ) + compared = [] + for position, (left, right) in enumerate(zip(before["rows"], after["rows"], strict=True)): + require( + left["position"] == right["position"] == position + and left["absolute_position"] == right["absolute_position"], + "saved replay positions differ", + ) + require( + left["target_rows"] == right["target_rows"] == target_rows, + "saved replay execution arms differ", + ) + compared.append(compare_rows(left["logits"], right["logits"])) + return compared, compare_rows(before["prefill"], after["prefill"]) + + +def compare_measurements(before, after): + """Join completed before/after evidence only for the identical frozen corpus.""" + for report in (before, after): + authenticate(report) + require( + report.get("schema") == "urn:qwen:d7-equivalence-summary:v1" + and report.get("status") == "MEASURED", + "before/after comparison requires completed native measurements", + ) + for key in ("corpus", "ordering", "scope"): + require(before[key] == after[key], "before/after measurement contract differs") + require(before["revision"] != after["revision"], "before and after are the same revision") + require( + before["metrics"]["positions"] == after["metrics"]["positions"] == 10000, + "before/after comparison requires all 10,000 positions in each measurement", + ) + return seal( + { + "schema": "urn:qwen:d7-equivalence-before-after:v1", + "before": before["sha256"], + "after": after["sha256"], + "corpus": before["corpus"], + "positions": 10000, + "top_k": { + str(k): { + "before": before["metrics"][str(k)], + "after": after["metrics"][str(k)], + "set_agreement_percentage_point_change": ( + after["metrics"][str(k)]["set_exact_percent"] + - before["metrics"][str(k)]["set_exact_percent"] + ), + } + for k in KS + }, + "scope": "observed native agreement; not a universal proof or answer-quality score", + } + ) + + +def choose_continuations(records, target): + """Select a fixed chronological corpus before inspecting any M1/M8 result. + + The prefill prediction is excluded. N saved output tokens provide N-1 decode + positions with a real pending successor; the last response may be truncated + in the replay corpus to make the evaluated-position budget exact. + """ + require( + type(target) is int and target > 0 and target % 8 == 0, + "position budget must contain complete M8 groups", + ) + selected, remaining, seen = [], target, set() + for row in records: + require( + row["mode"] == "full" and row["finish_reason"] == "stop", + "not a natural full-head response", + ) + require(row["trial"] not in seen, "duplicate saved response") + seen.add(row["trial"]) + # Never count a scheduler-shortened final batch as an M8 observation. + count = min(remaining, 8 * ((row["output_tokens"] - 1) // 8)) + if count > 0: + selected.append({**row, "evaluate_positions": count}) + remaining -= count + if remaining == 0: + break + require(remaining == 0, "saved natural responses do not cover the position budget") + return selected + + +class ReplaySchedule: + """Pure controller for the source-bound runner adapter; tests use no GPU.""" + + def __init__(self, prefix, output, *, speculation): + require(bool(prefix) and len(output) >= 2, "empty forced replay") + require(all(type(t) is int and t >= 0 for t in (*prefix, *output)), "invalid forced token") + self.prefix, self.output, self.speculation = prefix, output, speculation + self.cursor = 0 + self.prefill_done = False + self.prefill_cursor = 0 + + @property + def done(self): + return self.prefill_done and self.cursor == len(self.output) - 1 + + def expected_inputs(self, positions): + history = self.prefix + self.output + return [history[p] if p < len(history) else 0 for p in positions] + + def check_inputs(self, positions, inputs): + require(positions and len(positions) == len(inputs), "empty or incomplete model input") + require( + positions == list(range(positions[0], positions[0] + len(positions))), + "nonconsecutive model positions", + ) + require( + inputs == self.expected_inputs(positions), + "model did not consume the frozen token history", + ) + if self.prefill_done: + require(positions[0] == len(self.prefix) + self.cursor, "decode history is misaligned") + require( + len(positions) == (8 if self.speculation else 1), "wrong target execution width" + ) + else: + require(positions[0] == self.prefill_cursor, "prefill skipped or repeated a token") + require(positions[-1] < len(self.prefix), "prefill crossed into the output suffix") + self.prefill_cursor = positions[-1] + 1 + + def commit(self, drafts): + require(not self.done, "replay exceeded its exact position budget") + if not self.prefill_done: + require(drafts == 0, "speculation appeared during initial prefill") + require( + self.prefill_cursor == len(self.prefix), "prefill did not consume the full prompt" + ) + self.prefill_done = True + return { + "prefill": True, + "start": -1, + "count": 1, + "tokens": self.output[:1], + "reject": 0, + } + require(drafts == (7 if self.speculation else 0), "target width changed during replay") + count = min(drafts + 1, len(self.output) - 1 - self.cursor) + start = self.cursor + tokens = self.output[start + 1 : start + count + 1] + self.cursor += count + return { + "prefill": False, + "start": start, + "count": count, + "tokens": tokens, + "reject": drafts + 1 - count, + } + + def proposals(self): + return [ + self.output[p] if p < len(self.output) else 0 + for p in range(self.cursor + 1, self.cursor + 8) + ] + + def history_digest(self, position): + return digest(self.prefix + self.output[: position + 1]) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/conformance_transport.py b/benchmarks/conformance/src/qwen_r9700_lab/conformance_transport.py new file mode 100644 index 0000000..6401e8b --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/conformance_transport.py @@ -0,0 +1,453 @@ +"""Owned qualification processes and strict OpenAI/SSE transport. + +No GPU imports and no connection to a pre-existing inference endpoint. The +controller requires a fresh private workspace and an identity handshake from +its own child before sending a prompt. Logs and failed attempts are retained. +""" + +from __future__ import annotations + +import codecs +import contextlib +import json +import math +import os +import signal +import subprocess +import sys +import time +import urllib.error +import urllib.request +from pathlib import Path + +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, write_private + + +class ProtocolError(DiagnosticError): + pass + + +class BackendResponseError(ProtocolError): + """An explicit server error; never a timeout, disconnect or client failure.""" + + def __init__(self, message, *, http_status=None): + super().__init__(message) + self.http_status = http_status + + +def check_response_event(event): + if not isinstance(event, dict): + raise ProtocolError("backend response is not a JSON object") + if "error" in event: + error = event["error"] + if not isinstance(error, dict) or not isinstance(error.get("message"), str): + raise ProtocolError("malformed backend error object") + raise BackendResponseError("backend returned an explicit error") + + +class SSEDecoder: + """Incremental UTF-8 and SSE framing, including CRLF split across reads.""" + + def __init__(self): + self.decoder = codecs.getincrementaldecoder("utf-8")("strict") + self.buffer = "" + self.data = [] + self.done = False + + def feed(self, raw: bytes, *, final=False): + try: + self.buffer += self.decoder.decode(raw, final=final) + except UnicodeDecodeError as error: + raise ProtocolError("invalid UTF-8 in stream") from error + events = [] + while "\n" in self.buffer: + line, self.buffer = self.buffer.split("\n", 1) + line = line.removesuffix("\r") + if not line: + if self.data: + value = "\n".join(self.data) + self.data.clear() + if self.done: + raise ProtocolError("data after stream completion") + if value == "[DONE]": + self.done = True + else: + try: + event = json.loads(value) + except ValueError as error: + raise ProtocolError("invalid JSON in stream") from error + check_response_event(event) + events.append(event) + elif line.startswith("data:"): + self.data.append(line[5:].removeprefix(" ")) + elif line.startswith((":", "event:", "id:", "retry:")): + continue + else: + raise ProtocolError("unsupported stream field") + if final and (self.buffer or self.data or not self.done): + raise ProtocolError("stream ended without a complete DONE boundary") + return events + + +class Completion: + """Assemble one real response without inventing a finish or a tool call.""" + + def __init__(self): + self.content = "" + self.reasoning = "" + self.token_ids = [] + self.prompt_token_ids = None + self.tools = {} + self.finish = None + self.usage = None + self.events = 0 + + def _accept_prompt_ids(self, prompt_ids): + if prompt_ids is not None: + if not isinstance(prompt_ids, list) or any( + type(token) is not int or token < 0 for token in prompt_ids + ): + raise ProtocolError("invalid prompt token IDs") + if self.prompt_token_ids is not None and self.prompt_token_ids != prompt_ids: + raise ProtocolError("prompt token IDs changed during streaming") + self.prompt_token_ids = list(prompt_ids) + + def accept(self, event): + check_response_event(event) + self._accept_prompt_ids(event.get("prompt_token_ids")) + if event.get("usage") is not None: + self.usage = event["usage"] + choices = event.get("choices", []) + if len(choices) > 1: + raise ProtocolError("qualification expects exactly one choice") + if not choices: + return + choice = choices[0] + if choice.get("index", 0) != 0: + raise ProtocolError("unexpected choice index") + # The chat API places prompt IDs at response level; the completions API + # places them on its choice. Preserve both without allowing replacement. + self._accept_prompt_ids(choice.get("prompt_token_ids")) + delta = choice.get("delta", choice.get("message", {})) + if self.finish is not None: + raise ProtocolError("choice data after its finish boundary") + self.events += 1 + self.content += delta.get("content") or choice.get("text") or "" + self.reasoning += delta.get("reasoning") or delta.get("reasoning_content") or "" + ids = choice.get("token_ids") + if ids is not None: + if not isinstance(ids, list) or any(type(t) is not int or t < 0 for t in ids): + raise ProtocolError("invalid output token IDs") + self.token_ids.extend(ids) + for call in delta.get("tool_calls") or []: + index = call.get("index", len(self.tools) if "message" in choice else None) + if type(index) is not int or index < 0: + raise ProtocolError("missing tool delta index") + target = self.tools.setdefault(index, {"id": None, "name": "", "arguments": ""}) + if call.get("id"): + if target["id"] is not None and target["id"] != call["id"]: + raise ProtocolError("tool identity changed during streaming") + target["id"] = call["id"] + function = call.get("function", {}) + target["name"] += function.get("name") or "" + target["arguments"] += function.get("arguments") or "" + if choice.get("finish_reason") is not None: + self.finish = choice["finish_reason"] + + def result(self, *, allow_length=False): + if not self.events or self.finish not in {"stop", "tool_calls", "length"}: + raise ProtocolError("missing or unsupported finish reason") + if self.finish == "length" and not allow_length: + raise ProtocolError("generation exhausted its declared test budget") + if bool(self.tools) != (self.finish == "tool_calls"): + raise ProtocolError("tool emission and finish boundary disagree") + if sorted(self.tools) != list(range(len(self.tools))): + raise ProtocolError("tool indices have a gap") + tools = [] + for value in self.tools.values(): + if not value["id"] or not value["name"]: + raise ProtocolError("incomplete tool identity") + try: + args = json.loads(value["arguments"]) + except ValueError as error: + raise ProtocolError("unfinished tool arguments") from error + if not isinstance(args, dict): + raise ProtocolError("tool arguments must be an object") + tools.append({**value, "parsed_arguments": args}) + return { + "content": self.content, + "reasoning": self.reasoning, + "token_ids": self.token_ids, + "prompt_token_ids": self.prompt_token_ids, + "tools": tools, + "finish_reason": self.finish, + "usage": self.usage, + } + + +def live_group_members(group, *, proc_root=Path("/proc")): + """Read only the created process group, including surviving threads of a dead leader.""" + members = [] + for entry in proc_root.iterdir(): + if not entry.name.isdigit(): + continue + pid = int(entry.name) + try: + if os.getpgid(pid) != group: + continue + fields = (entry / "stat").read_text().rsplit(") ", 1)[1].split() + if int(fields[2]) != group: + continue + if fields[0] in {"Z", "X"}: + for task in (entry / "task").iterdir(): + try: + thread = (task / "stat").read_text().rsplit(") ", 1)[1].split() + except (FileNotFoundError, ProcessLookupError): + continue + if thread[0] not in {"Z", "X"}: + members.append( + { + "pid": pid, + "tid": int(task.name), + "state": thread[0], + "start_ticks": int(thread[19]), + } + ) + continue + members.append({"pid": pid, "state": fields[0], "start_ticks": int(fields[19])}) + except (FileNotFoundError, ProcessLookupError): + continue + except PermissionError: + members.append({"pid": pid, "state": "unreadable"}) + return sorted(members, key=lambda row: (row["pid"], row.get("tid", row["pid"]))) + + +def await_group_exit(group, *, timeout=2): + deadline = time.monotonic() + timeout + while members := live_group_members(group): + if time.monotonic() >= deadline: + return members + time.sleep(0.02) + return [] + + +class OwnedProcess: + def __init__(self, argv, root: Path, *, env, timeout): + if not argv or any(not isinstance(a, str) or "\0" in a for a in argv): + raise DiagnosticError("invalid qualification process arguments") + if timeout <= 0: + raise DiagnosticError("qualification requires a positive deadline") + root.mkdir(mode=0o700) + self.root, self.timeout, self.process = root, timeout, None + self.log = None + self.closed = False + self.close_error = None + self.started = time.monotonic() + write_private(root / "invocation.json", {"argv": argv, "deadline_seconds": timeout}) + try: + fd = os.open(root / "process.log", os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + self.log = os.fdopen(fd, "wb") + self.process = subprocess.Popen( + [ + sys.executable, + "-m", + "qwen_r9700_lab.conformance_supervisor", + str(os.getpid()), + str((root / "invocation.json").resolve()), + ], + env=env, + cwd=root, + stdin=subprocess.DEVNULL, + stdout=self.log, + stderr=self.log, + start_new_session=True, + ) + except BaseException: + self.close() + raise + + def check(self): + if self.process.poll() is not None: + raise DiagnosticError("owned qualification process exited; private log retained") + if time.monotonic() - self.started >= self.timeout: + raise TimeoutError("qualification process deadline expired") + + def wait(self): + try: + return self.process.wait( + timeout=max(0.001, self.timeout - (time.monotonic() - self.started)) + ) + except subprocess.TimeoutExpired as error: + raise TimeoutError("qualification child exceeded its deadline") from error + finally: + self.close() + + def close(self, *, crash=False, grace_seconds=10): + if self.closed: + if self.close_error is not None: + raise self.close_error + return + if ( + isinstance(grace_seconds, bool) + or not isinstance(grace_seconds, (int, float)) + or not math.isfinite(grace_seconds) + or grace_seconds <= 0 + ): + raise DiagnosticError("owned process requires a finite positive shutdown grace") + self.closed = True + members = [] + started = time.monotonic() + grace_expired = False + try: + if self.process is not None: + # Only the session we created; never pidof/pkill or a service manager. + with contextlib.suppress(ProcessLookupError): + if crash: + os.killpg(self.process.pid, signal.SIGKILL) + else: + os.kill(self.process.pid, signal.SIGTERM) + try: + self.process.wait(timeout=10 if crash else grace_seconds) + except subprocess.TimeoutExpired: + grace_expired = True + with contextlib.suppress(ProcessLookupError): + os.killpg(self.process.pid, signal.SIGKILL) + self.process.wait(timeout=10) + # A launcher may have exited before its owned workers did. + with contextlib.suppress(ProcessLookupError): + os.killpg(self.process.pid, signal.SIGKILL) + members = await_group_exit(self.process.pid) + if members: + raise DiagnosticError("owned workers remain alive after process-group shutdown") + except BaseException as error: + self.close_error = error + if self.process is not None: + from qwen_r9700_lab.conformance_gpu_lease import block_cleanup + + block_cleanup( + self.root, + process_group=self.process.pid, + reason=type(error).__name__ + ": " + str(error), + members=members, + ) + raise + finally: + if self.log is not None: + self.log.close() + write_private( + self.root / "shutdown.json", + { + "mode": "crash" if crash else "graceful", + "grace_seconds": grace_seconds, + "grace_expired": grace_expired, + "seconds": time.monotonic() - started, + "returncode": self.process.returncode if self.process is not None else None, + "cleanup_error": str(self.close_error) if self.close_error else None, + }, + ) + + def __enter__(self): + return self + + def __exit__(self, *_): + self.close() + + +class NoRedirects(urllib.request.HTTPRedirectHandler): + def redirect_request(self, *_args, **_kwargs): + raise ProtocolError("owned qualification endpoint attempted an HTTP redirect") + + +class OwnedClient: + def __init__(self, process: OwnedProcess, port: int, nonce: str, execution: str): + self.process, self.nonce, self.execution = process, nonce, execution + self.base = f"http://127.0.0.1:{port}" + self.verified = False + # Ignore HTTP_PROXY/ALL_PROXY: qualification never leaves loopback. + self.opener = urllib.request.build_opener(urllib.request.ProxyHandler({}), NoRedirects()) + + def request(self, path, body=None, *, timeout=30): + self.process.check() + if path != "/qwen-conformance/identity" and not self.verified: + raise DiagnosticError("owned server identity has not been verified") + if not path.startswith("/") or path.startswith("//"): + raise DiagnosticError("qualification requests require a local relative route") + request = urllib.request.Request( + self.base + path, + data=json.dumps(body, allow_nan=False).encode() if body is not None else None, + headers={"Content-Type": "application/json", "Authorization": "Bearer " + self.nonce}, + ) + return self.opener.open(request, timeout=timeout) + + def json(self, path, body=None, *, timeout=30): + with self.request(path, body, timeout=timeout) as response: + value = json.load(response) + if not isinstance(value, dict) or "error" in value: + raise ProtocolError("backend returned an error or malformed JSON") + return value + + def connect(self, *, timeout): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + self.process.check() + try: + identity = self.json("/qwen-conformance/identity", timeout=1) + except (urllib.error.URLError, TimeoutError): + time.sleep(0.05) + continue + if identity.get("nonce") != self.nonce or identity.get("execution") != self.execution: + raise DiagnosticError("server identity mismatch; no inference sent") + self.verified = True + return identity + raise TimeoutError("owned server did not become ready") + + def completion( + self, path, body, *, evidence: Path, allow_length=False, cancel_after=None, timeout=600 + ): + if not math.isfinite(timeout) or timeout <= 0: + raise DiagnosticError("completion requires a positive finite I/O timeout") + write_private(evidence.with_suffix(".request.json"), body) + fd = os.open(evidence.with_suffix(".response"), os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + assembled, decoder = Completion(), SSEDecoder() + started = time.monotonic() + try: + with os.fdopen(fd, "wb") as raw: + try: + response = self.request(path, body, timeout=timeout) + except urllib.error.HTTPError as error: + with error: + raw.write(error.read()) + raise BackendResponseError( + f"backend returned HTTP {error.code}", http_status=error.code + ) from error + with response: + if not body.get("stream", False): + data = response.read() + raw.write(data) + assembled.accept(json.loads(data)) + else: + while chunk := response.read1(4096): + self.process.check() + raw.write(chunk) + raw.flush() + for event in decoder.feed(chunk): + assembled.accept(event) + if cancel_after is not None and assembled.events >= cancel_after: + return {"cancelled": True, "observed_events": assembled.events} + decoder.feed(b"", final=True) + except Exception as error: + write_private( + evidence.with_suffix(".failure.json"), + { + "error_type": type(error).__name__, + "http_status": getattr(error, "http_status", None), + "elapsed_seconds": time.monotonic() - started, + "observed_events": assembled.events, + "response_bytes": evidence.with_suffix(".response").stat().st_size, + }, + ) + raise + result = assembled.result(allow_length=allow_length) + result["elapsed_seconds"] = time.monotonic() - started + write_private(evidence.with_suffix(".result.json"), result) + return result diff --git a/benchmarks/conformance/src/qwen_r9700_lab/diagnostic_contract.py b/benchmarks/conformance/src/qwen_r9700_lab/diagnostic_contract.py new file mode 100644 index 0000000..f8883f0 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/diagnostic_contract.py @@ -0,0 +1,504 @@ +"""Portable, offline contracts for inference diagnostics. + +An equal finite trace is evidence about the declared observations, not a proof +of model equivalence. Backend adapters own tensor/state extraction; this module +owns identities, coverage and comparison. It has no GPU or inference imports. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import re +import stat +from collections import Counter +from collections.abc import Iterable, Mapping +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Protocol + +SCHEMA = "urn:qwen:portable-diagnostic:v1" +SEMANTIC_FIELDS = frozenset( + { + "weights", + "weight_quantization", + "activation_quantization", + "attention", + "kv_representation", + "recurrence", + "position_encoding", + "tokenizer", + "chat_template", + "sampler", + "numerical_contract", + } +) +ARTIFACT_GROUPS = frozenset( + { + "source", + "generated_code", + "gpu_binary", + "compiler", + "runtime", + "hardware", + "model", + "tokenizer_template", + "configuration", + "adapter", + "reference", + } +) +SHA = re.compile(r"[0-9a-f]{64}\Z") +NAME = re.compile(r"[a-zA-Z0-9][a-zA-Z0-9_.:/-]{0,159}\Z") + + +class DiagnosticError(ValueError): + """Evidence is malformed, incomplete or not comparable.""" + + +def digest(value: Any) -> str: + # Same canonical JSON convention as the existing assurance trace writer. + data = json.dumps(value, allow_nan=False, sort_keys=True, separators=(",", ":")) + return hashlib.sha256(data.encode()).hexdigest() + + +def require_sha(value: object) -> str: + if not isinstance(value, str) or not SHA.fullmatch(value): + raise DiagnosticError("missing or invalid content identity") + return value + + +def require_name(value: object) -> str: + if not isinstance(value, str) or not NAME.fullmatch(value): + raise DiagnosticError("invalid diagnostic identifier") + return value + + +def integer(value: object, *, minimum: int = 0) -> int: + if type(value) is not int or value < minimum: + raise DiagnosticError("invalid diagnostic count or position") + return value + + +def semantic_identity(specification: Mapping[str, Any]) -> str: + """Changing an accepted approximation creates a different reference model.""" + if set(specification) != SEMANTIC_FIELDS: + raise DiagnosticError("reference semantics must explicitly describe every model component") + if any(not isinstance(v, dict) or not v for v in specification.values()): + raise DiagnosticError("reference components require explicit nonempty specifications") + return digest(specification) + + +def seal(document: Mapping[str, Any]) -> dict[str, Any]: + if "sha256" in document: + raise DiagnosticError("document is already sealed") + return {**document, "sha256": digest(document)} + + +def authenticate(document: Mapping[str, Any]) -> None: + unsigned = {k: v for k, v in document.items() if k != "sha256"} + if require_sha(document.get("sha256")) != digest(unsigned): + raise DiagnosticError("diagnostic document changed after publication") + + +def private_json(path: Path) -> dict[str, Any]: + """Read one stable owner-only file without following a final symlink.""" + descriptor = os.open(path, os.O_RDONLY | os.O_NOFOLLOW) + with os.fdopen(descriptor, "rb") as stream: + before = os.fstat(stream.fileno()) + if ( + not stat.S_ISREG(before.st_mode) + or before.st_uid != os.getuid() + or stat.S_IMODE(before.st_mode) & 0o077 + ): + raise DiagnosticError("diagnostic input must be an owned private regular file") + data = stream.read() + after = os.fstat(stream.fileno()) + if (before.st_size, before.st_mtime_ns, before.st_ctime_ns) != ( + after.st_size, + after.st_mtime_ns, + after.st_ctime_ns, + ): + raise DiagnosticError("diagnostic input changed during reading") + value = json.loads(data) + if not isinstance(value, dict): + raise DiagnosticError("diagnostic input must be an object") + return value + + +def write_private(path: Path, document: Mapping[str, Any]) -> None: + """Create once; incomplete writes never replace previous evidence.""" + data = (json.dumps(document, allow_nan=False, sort_keys=True, indent=2) + "\n").encode() + descriptor = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, 0o600) + with os.fdopen(descriptor, "wb") as stream: + stream.write(data) + stream.flush() + os.fsync(stream.fileno()) + descriptor = os.open(path.parent, os.O_RDONLY | os.O_DIRECTORY) + try: + os.fsync(descriptor) + finally: + os.close(descriptor) + + +def execution_manifest( + semantics: Mapping[str, Any], + files: Mapping[str, Mapping[str, Path]], + unavailable: Mapping[str, str], +) -> dict[str, Any]: + """Hash build artifacts once; requests refer to the resulting manifest hash. + + This records files supplied by the operator. A loaded-process attestation is + separate evidence; hashing a source checkout does not attest loaded kernels. + """ + from qwen_r9700_lab.manifests import sha256_file + + if set(files) | set(unavailable) != ARTIFACT_GROUPS or set(files) & set(unavailable): + raise DiagnosticError("every artifact group must be present or explicitly unavailable") + if any(not isinstance(reason, str) or not reason.strip() for reason in unavailable.values()): + raise DiagnosticError("unavailable artifacts need a reason") + artifacts = [] + for group, members in sorted(files.items()): + if not members: + raise DiagnosticError("an empty artifact group does not establish coverage") + for name, path in sorted(members.items()): + require_name(name) + path = Path(path) + before = path.stat() + if not stat.S_ISREG(before.st_mode): + raise DiagnosticError("artifact is not a regular file") + identity = sha256_file(path) + after = path.stat() + fields = ("st_dev", "st_ino", "st_size", "st_mtime_ns", "st_ctime_ns") + if any(getattr(before, key) != getattr(after, key) for key in fields): + raise DiagnosticError("artifact changed during hashing") + artifacts.append( + {"group": group, "name": name, "sha256": identity, "bytes": after.st_size} + ) + return seal( + { + "schema": SCHEMA + "/execution", + "semantics_sha256": semantic_identity(semantics), + "artifacts": artifacts, + "unavailable": dict(unavailable), + "artifact_inventory_complete": not unavailable, + "loaded_process_attested": False, + "formal_equivalence_proven": False, + } + ) + + +def verify_source_bindings(root: Path, expected: Mapping[str, str]) -> None: + """Authenticate all native bindings before an adapter may install any hook.""" + from qwen_r9700_lab.manifests import sha256_file + + if not expected: + raise DiagnosticError("native adapter has no source bindings") + root = root.resolve(strict=True) + for relative, expected_hash in expected.items(): + path = (root / relative).resolve(strict=True) + if Path(relative).is_absolute() or not path.is_relative_to(root): + raise DiagnosticError("native binding escapes its declared runtime") + if sha256_file(path) != require_sha(expected_hash): + raise DiagnosticError("native source binding changed; adapter needs requalification") + + +@dataclass(frozen=True, order=True) +class Boundary: + layer: int + name: str + kind: str = "tensor" + + def __post_init__(self) -> None: + integer(self.layer) + require_name(self.name) + if self.kind not in {"tensor", "logical_state"}: + raise DiagnosticError("unknown boundary value kind") + + def document(self) -> dict[str, Any]: + return {"layer": self.layer, "name": self.name, "kind": self.kind} + + +@dataclass(frozen=True) +class Observation: + position: int + boundary: Boundary + value_sha256: str + nonfinite: int | None = None + + def document(self) -> dict[str, Any]: + return { + "position": integer(self.position), + **self.boundary.document(), + "value_sha256": require_sha(self.value_sha256), + "nonfinite": None if self.nonfinite is None else integer(self.nonfinite), + } + + +def trace( + *, + execution_sha256: str, + semantics_sha256: str, + adapter_sha256: str, + input_sha256: str, + mode: str, + positions: list[int], + boundaries: list[Boundary], + observations: Iterable[Observation], +) -> dict[str, Any]: + if mode not in {"forced_tokens", "prefill", "snapshot_roundtrip"}: + raise DiagnosticError("free generation is not forced-token differential execution") + if not positions or positions != sorted(set(positions)): + raise DiagnosticError("capture positions must be nonempty, unique and ordered") + for position in positions: + integer(position) + if not boundaries or len(set(boundaries)) != len(boundaries): + raise DiagnosticError("capture boundaries must be nonempty and unique") + rows = [observation.document() for observation in observations] + expected = {(p, b.layer, b.name, b.kind) for p in positions for b in boundaries} + keys = [(r["position"], r["layer"], r["name"], r["kind"]) for r in rows] + if len(keys) != len(set(keys)) or set(keys) != expected: + raise DiagnosticError("capture has missing, duplicate or undeclared observations") + order = {(b.layer, b.name, b.kind): i for i, b in enumerate(boundaries)} + rows.sort(key=lambda r: (r["position"], order[r["layer"], r["name"], r["kind"]])) + return seal( + { + "schema": SCHEMA + "/trace", + "execution_sha256": require_sha(execution_sha256), + "semantics_sha256": require_sha(semantics_sha256), + "adapter_sha256": require_sha(adapter_sha256), + "input_sha256": require_sha(input_sha256), + "mode": mode, + "positions": positions, + "boundaries": [b.document() for b in boundaries], + "observations": rows, + } + ) + + +def validate_trace(value: Mapping[str, Any]) -> None: + authenticate(value) + if value.get("schema") != SCHEMA + "/trace": + raise DiagnosticError("unsupported trace schema") + rebuilt = trace( + **{ + key: value[key] + for key in ( + "execution_sha256", + "semantics_sha256", + "adapter_sha256", + "input_sha256", + "mode", + "positions", + ) + }, + boundaries=[Boundary(**b) for b in value["boundaries"]], + observations=[ + Observation( + r["position"], + Boundary(r["layer"], r["name"], r["kind"]), + r["value_sha256"], + r["nonfinite"], + ) + for r in value["observations"] + ], + ) + if rebuilt != value: + raise DiagnosticError("trace contains unknown fields or noncanonical observations") + + +def compare_traces(left: Mapping[str, Any], right: Mapping[str, Any]) -> dict[str, Any]: + """Compare declared boundaries, not backend-specific addresses or row slots.""" + for value in (left, right): + validate_trace(value) + for key in ("semantics_sha256", "input_sha256", "mode", "positions", "boundaries"): + if left[key] != right[key]: + raise DiagnosticError("traces differ in reference semantics, input or coverage") + differences = [] + for a, b in zip(left["observations"], right["observations"], strict=True): + if a["value_sha256"] != b["value_sha256"] or a["nonfinite"] or b["nonfinite"]: + differences.append({key: a[key] for key in ("position", "layer", "name", "kind")}) + return seal( + { + "schema": SCHEMA + "/comparison", + "left_sha256": left["sha256"], + "right_sha256": right["sha256"], + "observations": len(left["observations"]), + "equal_observed_boundaries": not differences, + "differing_boundaries": len(differences), + "first_difference": differences[0] if differences else None, + "claim": "finite_declared_boundary_comparison", + "finite_values_verified": all( + r["nonfinite"] == 0 for value in (left, right) for r in value["observations"] + ), + "numerical_difference_is_automatically_a_bug": False, + "sampling_distribution_verified": False, + "formal_equivalence_proven": False, + } + ) + + +def logical_cache_state( + allocations: list[dict[str, Any]], + mappings: list[dict[str, Any]], + free: list[int], +) -> list[dict[str, Any]]: + """Validate each allocator locally, compare logical ownership across backends. + + Shared immutable prefix blocks and different physical block numbering are + legal. Adapters must export the same logical granularity and include all + references, including pins and other references not held by a sequence. + """ + physical = {} + for block in allocations: + if set(block) != {"id", "refcount", "external_references"}: + raise DiagnosticError("allocation descriptor is incomplete") + key = integer(block["id"]) + if key in physical: + raise DiagnosticError("duplicate physical allocation") + physical[key] = block + free_set = {integer(value) for value in free} + if len(free_set) != len(free) or free_set & physical.keys(): + raise DiagnosticError("free and allocated blocks overlap or repeat") + references: Counter[int] = Counter() + logical = [] + identities = set() + for row in mappings: + if set(row) != {"sequence", "component", "layer", "start", "end", "block", "sha256"}: + raise DiagnosticError("logical cache mapping is incomplete") + key = integer(row["block"]) + if key not in physical: + raise DiagnosticError("live sequence reaches an unallocated or free block") + integer(row["layer"]) + start, end = integer(row["start"]), integer(row["end"]) + if end <= start: + raise DiagnosticError("cache mapping has an empty or reversed token span") + identity = ( + require_name(row["sequence"]), + require_name(row["component"]), + row["layer"], + start, + end, + ) + if identity in identities: + raise DiagnosticError("duplicate logical cache ownership") + identities.add(identity) + references[key] += 1 + logical.append( + { + **{k: row[k] for k in ("sequence", "component", "layer", "start", "end")}, + "sha256": require_sha(row["sha256"]), + } + ) + for key, block in physical.items(): + count = integer(block["refcount"], minimum=1) + if count != references[key] + integer(block["external_references"]): + raise DiagnosticError("allocator reference count differs from live references") + ordered = sorted(logical, key=lambda r: (r["sequence"], r["component"], r["layer"], r["start"])) + previous = {} + for row in ordered: + group = (row["sequence"], row["component"], row["layer"]) + if row["start"] < previous.get(group, 0): + raise DiagnosticError("logical cache ownership overlaps within one sequence") + previous[group] = row["end"] + return ordered + + +def remaining_materialized(before, old_pending, emitted, new_pending): + """Shared exact-integer accounting, also executed symbolically by Z3.""" + return before + old_pending + emitted - new_pending + + +def validate_speculative_commit( + *, + before_materialized: int, + before_pending: int, + drafted: int, + accepted: int, + emitted: int, + after_materialized: int, + after_pending: int, + component_versions: Mapping[str, int], + required_components: set[str], +) -> None: + """Check one successful publication using explicit pending-token accounting. + + Each version is the materialized prefix length represented by that state, + not an implementation's physical slot/generation number. Rejected suffix + independence additionally requires comparing the actual state values against + serial replay; these integer invariants alone do not establish it. + """ + for value in ( + before_materialized, + before_pending, + drafted, + accepted, + emitted, + after_materialized, + after_pending, + ): + integer(value) + if before_pending > 1 or after_pending > 1 or accepted > drafted or emitted != accepted + 1: + raise DiagnosticError("invalid speculative acceptance or pending-token accounting") + if after_materialized != remaining_materialized( + before_materialized, before_pending, emitted, after_pending + ): + raise DiagnosticError("materialized and emitted token counts disagree") + if not required_components or set(component_versions) != required_components: + raise DiagnosticError("persistent state component coverage is incomplete") + for name, version in component_versions.items(): + require_name(name) + if integer(version) != after_materialized: + raise DiagnosticError("persistent state version includes rejected or stale tokens") + + +class ForcedTokenAdapter(Protocol): + """A native adapter must consume the supplied token; never its own argmax. + + reset/step return only after the observed state transition is complete. + step's position is the token *consumed*, not the sampled pending bonus token. + A backend requiring another materialization convention needs an explicit + adapter, not an off-by-one exception in the common comparator. + """ + + execution_sha256: str + semantics_sha256: str + adapter_sha256: str + + def reset(self, prefix: tuple[int, ...]) -> None: ... + def step(self, token: int, position: int) -> Iterable[Observation]: ... + def close(self) -> None: ... + + +def forced_token_replay( + adapter: ForcedTokenAdapter, + prefix: tuple[int, ...], + suffix: tuple[int, ...], + boundaries: list[Boundary], +) -> dict[str, Any]: + """Reusable driver; private tokens never appear in its evidence document.""" + if not prefix or not suffix: + raise DiagnosticError("forced replay requires a prefix and a nonempty suffix") + for token in (*prefix, *suffix): + integer(token) + observations = [] + try: + adapter.reset(prefix) + for position, token in enumerate(suffix, start=len(prefix)): + rows = list(adapter.step(token, position)) + if any(row.position != position for row in rows): + raise DiagnosticError("adapter returned state for a different consumed token") + observations.extend(rows) + finally: + adapter.close() + return trace( + execution_sha256=adapter.execution_sha256, + semantics_sha256=adapter.semantics_sha256, + adapter_sha256=adapter.adapter_sha256, + input_sha256=digest({"prefix": prefix, "forced_suffix": suffix}), + mode="forced_tokens", + positions=list(range(len(prefix), len(prefix) + len(suffix))), + boundaries=boundaries, + observations=observations, + ) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/exact_fp8_metrics.py b/benchmarks/conformance/src/qwen_r9700_lab/exact_fp8_metrics.py new file mode 100644 index 0000000..2743d23 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/exact_fp8_metrics.py @@ -0,0 +1,77 @@ +"""Exact, bounded integer diagnostics for pairs of E4M3 cache bytes. + +This does not decide tensor equality or authenticate files. Those checks remain +the caller's responsibility. The histogram replaces repeated floating-point +expansion only for the diagnostic sums, with an explicit overflow bound. +""" + +from __future__ import annotations + +from functools import lru_cache + +import numpy as np + +MAX_ELEMENTS = 16 * 1024**2 +DTYPES = frozenset({"fp8_e4m3fn", "fp8_e4m3fnuz"}) + + +@lru_cache(maxsize=2) +def tables(dtype): + if dtype not in DTYPES: + raise ValueError("unsupported FP8 representation") + # Decode independently in integer units: FN's unit is 2^-9 and FNUZ's + # is 2^-10. Both have the same integer exponent/significand expression. + code = np.arange(256, dtype=np.int64) + exponent, mantissa = (code >> 3) & 15, code & 7 + units = np.where(exponent == 0, mantissa, (8 + mantissa) << np.maximum(exponent - 1, 0)) + units = np.where(code & 128, -units, units) + finite = code != 128 if dtype.endswith("fnuz") else ~((exponent == 15) & (mantissa == 7)) + valid = finite[:, None] & finite[None, :] + absolute = np.where(valid, np.abs(units[:, None] - units[None, :]), 0).reshape(-1) + square = absolute * absolute + ref_square = np.where(valid, units[:, None] ** 2, 0).reshape(-1) + if int(square.max()) * MAX_ELEMENTS > np.iinfo(np.int64).max: + raise ValueError("FP8 integer accumulation bound is insufficient") + result = { + "absolute": absolute, + "square": square, + "reference_square": ref_square, + "valid": valid.reshape(-1).astype(np.int64), + "nonfinite_a": np.repeat(~finite, 256).astype(np.int64), + "nonfinite_b": np.tile(~finite, 256).astype(np.int64), + "scale": 1024 if dtype.endswith("fnuz") else 512, + } + for value in result.values(): + if isinstance(value, np.ndarray): + value.flags.writeable = False + return result + + +def chunk_metrics(left: bytes, right: bytes, dtype: str) -> dict: + """Match the existing finite-pair long-double diagnostic sums exactly. + + At most 2^24 elements, each squared difference at most 491520^2 integer + units, gives a nonnegative sum below 2^62. Conversion to an extended + significand of at least 64 bits is exact. Power-of-two scaling is exact. + """ + if len(left) != len(right) or len(left) > MAX_ELEMENTS: + raise ValueError("FP8 metric chunk lengths differ or exceed the proved integer bound") + if np.finfo(np.longdouble).nmant < 63: + raise ValueError("exact FP8 metrics require at least a 64-bit significand") + table = tables(dtype) + a, b = np.frombuffer(left, np.uint8), np.frombuffer(right, np.uint8) + pair = (a.astype(np.uint16) << 8) | b + histogram = np.bincount(pair, minlength=65536).astype(np.int64, copy=False) + scale = table["scale"] + present = histogram != 0 + maximum = int(table["absolute"][present].max()) if present.any() else 0 + return { + "max_abs": np.longdouble(maximum) / scale, + "squared": np.longdouble(histogram @ table["square"]) / (scale * scale), + "reference_squared": np.longdouble(histogram @ table["reference_square"]) / (scale * scale), + "count": int(histogram @ table["valid"]), + "nonfinite": [ + int(histogram @ table["nonfinite_a"]), + int(histogram @ table["nonfinite_b"]), + ], + } diff --git a/benchmarks/conformance/src/qwen_r9700_lab/manifests.py b/benchmarks/conformance/src/qwen_r9700_lab/manifests.py new file mode 100644 index 0000000..94a5186 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/manifests.py @@ -0,0 +1,671 @@ +"""Capture content-addressed, create-once evidence manifests.""" + +from __future__ import annotations + +import fnmatch +import hashlib +import json +import os +import platform +import shutil +import socket +import subprocess +import sys +import tempfile +from contextlib import suppress +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + +from qwen_r9700_lab import __version__ +from qwen_r9700_lab.config import ( + ConfigurationError, + find_project_root, + load_json_object, + validate_config, + validate_instance, + validate_profile_semantics, +) + +SCHEMA_IDS = { + "environment": "urn:qwen-r9700-lab:schema:environment-manifest:v1", + "model": "urn:qwen-r9700-lab:schema:model-manifest:v1", + "run": "urn:qwen-r9700-lab:schema:run-manifest:v1", +} + +RELEVANT_ENVIRONMENT_VARIABLES = ( + "AMD_LOG_LEVEL", + "GGML_VK_ALLOW_GRAPHICS_QUEUE", + "GPU_DEVICE_ORDINAL", + "HIP_LAUNCH_BLOCKING", + "HIP_VISIBLE_DEVICES", + "HSA_OVERRIDE_GFX_VERSION", + "PYTORCH_ROCM_ARCH", + "ROCM_PATH", + "ROCR_VISIBLE_DEVICES", + "VLLM_ALLOW_LONG_MAX_MODEL_LEN", + "VULKAN_SDK", +) + + +def utc_now() -> str: + """Return a compact UTC timestamp suitable for evidence records.""" + + return datetime.now(UTC).isoformat(timespec="seconds").replace("+00:00", "Z") + + +def canonical_bytes(value: Any) -> bytes: + """Encode JSON canonically for stable content hashes.""" + + return json.dumps( + value, + ensure_ascii=False, + separators=(",", ":"), + sort_keys=True, + ).encode("utf-8") + + +def sha256_file(path: Path) -> str: + """Hash one file without loading it into memory.""" + + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(8 * 1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _manifest_with_id(document: dict[str, Any]) -> dict[str, Any]: + if "manifest_id" in document: + raise ConfigurationError("manifest payload already contains manifest_id") + result = dict(document) + result["manifest_id"] = f"sha256:{hashlib.sha256(canonical_bytes(document)).hexdigest()}" + return result + + +def write_immutable_manifest( + output: Path, + document: dict[str, Any], + schema_name: str, + project_root: Path | None = None, +) -> dict[str, Any]: + """Validate and atomically link a read-only manifest without replacing a path.""" + + root = project_root or find_project_root(output.parent) + completed = _manifest_with_id(document) + validate_instance(completed, root / "schemas" / schema_name, str(output)) + + # Keep the final path itself unresolved so a pre-existing symlink is rejected by the + # no-replace hard link instead of being followed to its target. + output = output.expanduser().absolute() + output.parent.mkdir(parents=True, exist_ok=True) + payload = json.dumps(completed, ensure_ascii=False, indent=2, sort_keys=True) + "\n" + + temporary_path: Path | None = None + try: + with tempfile.NamedTemporaryFile( + mode="w", + encoding="utf-8", + dir=output.parent, + prefix=".qwen-r9700-manifest-", + delete=False, + ) as handle: + temporary_path = Path(handle.name) + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + temporary_path.chmod(0o444) + try: + os.link(temporary_path, output) + except FileExistsError as error: + raise ConfigurationError(f"refusing to replace existing manifest: {output}") from error + directory_fd = os.open(output.parent, os.O_RDONLY | os.O_DIRECTORY) + try: + os.fsync(directory_fd) + finally: + os.close(directory_fd) + finally: + if temporary_path is not None: + temporary_path.unlink(missing_ok=True) + return completed + + +def _read_os_release() -> dict[str, str]: + result: dict[str, str] = {} + path = Path("/etc/os-release") + if not path.is_file(): + return result + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if "=" not in line or line.startswith("#"): + continue + key, value = line.split("=", 1) + result[key] = value.strip().strip('"') + return result + + +def _cpu_model() -> str | None: + path = Path("/proc/cpuinfo") + if not path.is_file(): + return platform.processor() or None + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if line.lower().startswith("model name") and ":" in line: + return line.split(":", 1)[1].strip() + return platform.processor() or None + + +def _memory_total_bytes() -> int | None: + path = Path("/proc/meminfo") + if not path.is_file(): + return None + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if line.startswith("MemTotal:"): + return int(line.split()[1]) * 1024 + return None + + +def _probe(argv: list[str], timeout_seconds: float = 10.0, limit: int = 65_536) -> dict[str, Any]: + executable = shutil.which(argv[0]) + if executable is None: + return {"argv": argv, "status": "unavailable"} + try: + process = subprocess.run( + [executable, *argv[1:]], + check=False, + capture_output=True, + text=True, + timeout=timeout_seconds, + ) + except subprocess.TimeoutExpired as error: + stdout = ( + error.stdout.decode(errors="replace") + if isinstance(error.stdout, bytes) + else error.stdout + ) + stderr = ( + error.stderr.decode(errors="replace") + if isinstance(error.stderr, bytes) + else error.stderr + ) + return { + "argv": argv, + "status": "timeout", + "stdout": (stdout or "")[:limit], + "stderr": (stderr or "")[:limit], + } + + stdout = process.stdout[:limit] + stderr = process.stderr[:limit] + return { + "argv": argv, + "status": "completed", + "returncode": process.returncode, + "stdout": stdout, + "stderr": stderr, + "stdout_truncated": len(process.stdout) > limit, + "stderr_truncated": len(process.stderr) > limit, + } + + +def _read_sysfs_text(path: Path) -> str | None: + """Read one small sysfs attribute, preserving absence as null evidence.""" + + try: + value = path.read_text(encoding="utf-8", errors="replace").strip() + except OSError: + return None + return value or None + + +def _read_sysfs_int(path: Path) -> int | None: + value = _read_sysfs_text(path) + if value is None: + return None + try: + return int(value, 10) + except ValueError: + return None + + +def _resource0_evidence(device_path: Path) -> dict[str, Any]: + byte_size: int | None = None + with suppress(OSError): + byte_size = (device_path / "resource0").stat().st_size + + table_entry = None + resource_table = _read_sysfs_text(device_path / "resource") + if resource_table is not None: + table_entry = resource_table.splitlines()[0] + + range_start = None + range_end = None + flags = None + range_size_bytes: int | None = None + if table_entry is not None: + fields = table_entry.split() + if len(fields) >= 3: + try: + start = int(fields[0], 16) + end = int(fields[1], 16) + except ValueError: + pass + else: + range_start = f"0x{start:016x}" + range_end = f"0x{end:016x}" + flags = fields[2] + if start == 0 and end == 0: + range_size_bytes = 0 + elif end >= start: + range_size_bytes = end - start + 1 + + sizes_match = None + if byte_size is not None and range_size_bytes is not None: + sizes_match = byte_size == range_size_bytes + return { + "byte_size": byte_size, + "resource_table_entry": table_entry, + "range_start": range_start, + "range_end": range_end, + "range_flags": flags, + "range_size_bytes": range_size_bytes, + "sizes_match": sizes_match, + } + + +def _drm_node_type(name: str) -> str | None: + for prefix, node_type in ( + ("card", "primary"), + ("renderD", "render"), + ("controlD", "control"), + ): + suffix = name.removeprefix(prefix) + if suffix != name and suffix.isdigit(): + return node_type + return None + + +def _drm_nodes(device_path: Path, dev_dri_root: Path) -> list[dict[str, Any]]: + drm_path = device_path / "drm" + try: + candidates = sorted(drm_path.iterdir(), key=lambda path: path.name) + except OSError: + return [] + + nodes: list[dict[str, Any]] = [] + for candidate in candidates: + node_type = _drm_node_type(candidate.name) + if node_type is None: + continue + device_node = dev_dri_root / candidate.name + nodes.append( + { + "name": candidate.name, + "node_type": node_type, + "major_minor": _read_sysfs_text(candidate / "dev"), + "device_path": str(device_node), + "device_path_exists": device_node.exists(), + } + ) + return nodes + + +def _driver_name(device_path: Path) -> str | None: + try: + return (device_path / "driver").resolve(strict=True).name + except (OSError, RuntimeError): + return None + + +def _lspci_region_0(pci_bdf: str) -> tuple[str | None, dict[str, Any]]: + probe = _probe(["lspci", "-s", pci_bdf, "-vv"]) + region_0 = None + if probe.get("status") == "completed": + for line in probe.get("stdout", "").splitlines(): + stripped = line.strip() + if stripped.startswith("Region 0:"): + region_0 = stripped + break + return region_0, probe + + +def _amd_gpu_evidence( + device_path: Path, + pci_bdf: str, + class_code: str, + vendor_id: str, + device_id: str | None, + dev_dri_root: Path, +) -> dict[str, Any]: + region_0, lspci_probe = _lspci_region_0(pci_bdf) + return { + "pci_bdf": pci_bdf, + "class_code": class_code, + "vendor_id": vendor_id, + "device_id": device_id, + "driver": _driver_name(device_path), + "lspci_region_0": region_0, + "lspci_probe": lspci_probe, + "resource0": _resource0_evidence(device_path), + "vram": { + "total_bytes": _read_sysfs_int(device_path / "mem_info_vram_total"), + "used_bytes": _read_sysfs_int(device_path / "mem_info_vram_used"), + }, + "pcie_link": { + "current_speed": _read_sysfs_text(device_path / "current_link_speed"), + "current_width": _read_sysfs_text(device_path / "current_link_width"), + "maximum_speed": _read_sysfs_text(device_path / "max_link_speed"), + "maximum_width": _read_sysfs_text(device_path / "max_link_width"), + }, + "power": { + "control": _read_sysfs_text(device_path / "power" / "control"), + "runtime_status": _read_sysfs_text(device_path / "power" / "runtime_status"), + }, + "drm_nodes": _drm_nodes(device_path, dev_dri_root), + } + + +def collect_pci_display_inventory( + pci_devices_root: Path = Path("/sys/bus/pci/devices"), + dev_dri_root: Path = Path("/dev/dri"), +) -> dict[str, Any]: + """Enumerate display-class PCI devices and collect per-BDF AMD GPU evidence.""" + + try: + candidates = sorted(pci_devices_root.iterdir(), key=lambda path: path.name) + except FileNotFoundError: + return { + "status": "unavailable", + "sysfs_pci_devices_path": str(pci_devices_root), + "display_device_count": 0, + "amd_gpu_count": 0, + "display_devices": [], + "amd_gpus": [], + } + except OSError as error: + return { + "status": "error", + "error": str(error), + "sysfs_pci_devices_path": str(pci_devices_root), + "display_device_count": 0, + "amd_gpu_count": 0, + "display_devices": [], + "amd_gpus": [], + } + + display_devices: list[dict[str, Any]] = [] + amd_gpus: list[dict[str, Any]] = [] + for device_path in candidates: + class_code = _read_sysfs_text(device_path / "class") + if class_code is None: + continue + try: + is_display = int(class_code, 16) >> 16 == 0x03 + except ValueError: + continue + if not is_display: + continue + + class_code = class_code.lower() + vendor_id_value = _read_sysfs_text(device_path / "vendor") + device_id_value = _read_sysfs_text(device_path / "device") + vendor_id = vendor_id_value.lower() if vendor_id_value is not None else None + device_id = device_id_value.lower() if device_id_value is not None else None + display_devices.append( + { + "pci_bdf": device_path.name, + "class_code": class_code, + "vendor_id": vendor_id, + "device_id": device_id, + "driver": _driver_name(device_path), + } + ) + if vendor_id is not None and vendor_id.lower() == "0x1002": + amd_gpus.append( + _amd_gpu_evidence( + device_path, + device_path.name, + class_code, + vendor_id, + device_id, + dev_dri_root, + ) + ) + + return { + "status": "captured", + "sysfs_pci_devices_path": str(pci_devices_root), + "display_device_count": len(display_devices), + "amd_gpu_count": len(amd_gpus), + "display_devices": display_devices, + "amd_gpus": amd_gpus, + } + + +def capture_environment( + hardware_config: Path, + output: Path, + project_root: Path | None = None, + *, + pci_devices_root: Path = Path("/sys/bus/pci/devices"), + dev_dri_root: Path = Path("/dev/dri"), +) -> dict[str, Any]: + """Capture host identity, runtime versions, and bounded GPU-stack probes.""" + + root = project_root or find_project_root(hardware_config.parent) + hardware = validate_config(hardware_config, root) + if hardware["kind"] != "hardware": + raise ConfigurationError(f"{hardware_config}: expected kind 'hardware'") + + document: dict[str, Any] = { + "$schema": SCHEMA_IDS["environment"], + "schema_version": 1, + "manifest_type": "environment", + "captured_at": utc_now(), + "collector": {"name": "qwen-r9700-lab", "version": __version__}, + "expected_hardware": { + "config_path": str(hardware_config.expanduser().resolve()), + "config_sha256": sha256_file(hardware_config), + "config": hardware, + }, + "host": { + "hostname": socket.gethostname(), + "machine": platform.machine(), + "kernel_release": platform.release(), + "kernel_version": platform.version(), + "operating_system": platform.system(), + "os_release": _read_os_release(), + "cpu_model": _cpu_model(), + "logical_cpu_count": os.cpu_count(), + "memory_total_bytes": _memory_total_bytes(), + }, + "runtime": { + "python_executable": sys.executable, + "python_version": platform.python_version(), + "python_implementation": platform.python_implementation(), + "relevant_environment": { + name: os.environ.get(name) for name in RELEVANT_ENVIRONMENT_VARIABLES + }, + }, + "pci_display_inventory": collect_pci_display_inventory(pci_devices_root, dev_dri_root), + "probes": { + "uv": _probe(["uv", "--version"]), + "pci": _probe(["lspci", "-Dnnk"]), + "rocm_info": _probe(["rocminfo"]), + "rocm_smi": _probe( + ["rocm-smi", "--showproductname", "--showdriverversion", "--showmeminfo", "vram"] + ), + "amd_smi": _probe(["amd-smi", "version"]), + "vulkan": _probe(["vulkaninfo", "--summary"]), + "glslc": _probe(["glslc", "--version"]), + "hipcc": _probe(["hipcc", "--version"]), + }, + } + return write_immutable_manifest( + output, + document, + "environment-manifest.schema.json", + root, + ) + + +def _is_excluded(relative_path: str, patterns: list[str]) -> bool: + return any(fnmatch.fnmatch(relative_path, pattern) for pattern in patterns) + + +def _model_files(model_dir: Path, exclude: list[str]) -> list[dict[str, Any]]: + files: list[dict[str, Any]] = [] + for path in sorted(model_dir.rglob("*")): + relative = path.relative_to(model_dir).as_posix() + if _is_excluded(relative, exclude): + continue + if path.is_dir(): + continue + if not path.is_file(): + raise ConfigurationError(f"model tree contains a non-file entry: {path}") + record: dict[str, Any] = { + "path": relative, + "size_bytes": path.stat().st_size, + "sha256": sha256_file(path), + } + if path.is_symlink(): + record["symlink_target"] = str(path.readlink()) + files.append(record) + return files + + +def capture_model( + model_config: Path, + model_dir: Path, + output: Path, + project_root: Path | None = None, +) -> dict[str, Any]: + """Hash a local model tree and bind it to a checked-in model declaration.""" + + root = project_root or find_project_root(model_config.parent) + configuration = validate_config(model_config, root) + if configuration["kind"] != "model": + raise ConfigurationError(f"{model_config}: expected kind 'model'") + + try: + resolved_model_dir = model_dir.expanduser().resolve(strict=True) + except FileNotFoundError as error: + raise ConfigurationError(f"model directory does not exist: {model_dir}") from error + if not resolved_model_dir.is_dir(): + raise ConfigurationError(f"model path is not a directory: {resolved_model_dir}") + + files = _model_files(resolved_model_dir, configuration["manifest_exclude"]) + if not files: + raise ConfigurationError( + f"model directory contains no included files: {resolved_model_dir}" + ) + tree_sha256 = hashlib.sha256(canonical_bytes(files)).hexdigest() + + document: dict[str, Any] = { + "$schema": SCHEMA_IDS["model"], + "schema_version": 1, + "manifest_type": "model", + "captured_at": utc_now(), + "collector": {"name": "qwen-r9700-lab", "version": __version__}, + "configuration": { + "config_path": str(model_config.expanduser().resolve()), + "config_sha256": sha256_file(model_config), + "config": configuration, + }, + "model_tree": { + "directory": str(resolved_model_dir), + "file_count": len(files), + "total_size_bytes": sum(record["size_bytes"] for record in files), + "tree_sha256": tree_sha256, + "files": files, + }, + } + return write_immutable_manifest(output, document, "model-manifest.schema.json", root) + + +def load_manifest(path: Path, schema_name: str, project_root: Path) -> dict[str, Any]: + """Load a manifest and verify both schema and embedded content address.""" + + manifest = load_json_object(path) + validate_instance(manifest, project_root / "schemas" / schema_name, str(path)) + expected_id = manifest["manifest_id"] + payload = dict(manifest) + del payload["manifest_id"] + actual_id = f"sha256:{hashlib.sha256(canonical_bytes(payload)).hexdigest()}" + if expected_id != actual_id: + raise ConfigurationError( + f"{path}: manifest_id mismatch (expected {actual_id}, found {expected_id})" + ) + return manifest + + +def prepare_run( + environment_manifest: Path, + model_manifest: Path, + engine_config: Path, + profile_config: Path, + reasoning_effort: str, + preserve_thinking: bool, + output: Path, + project_root: Path | None = None, +) -> dict[str, Any]: + """Bind immutable evidence and orthogonal policy choices into a run declaration.""" + + root = project_root or find_project_root(engine_config.parent) + environment = load_manifest( + environment_manifest, + "environment-manifest.schema.json", + root, + ) + model = load_manifest(model_manifest, "model-manifest.schema.json", root) + engine = validate_config(engine_config, root) + profile = validate_config(profile_config, root) + if engine["kind"] != "engine": + raise ConfigurationError(f"{engine_config}: expected kind 'engine'") + if profile["kind"] != "profile": + raise ConfigurationError(f"{profile_config}: expected kind 'profile'") + validate_profile_semantics(profile, profile_config) + + model_format = model["configuration"]["config"]["format"] + if model_format not in engine["supported_model_formats"]: + raise ConfigurationError( + f"engine {engine['id']!r} does not declare support for model format {model_format!r}" + ) + + thinking_enabled = reasoning_effort != "off" + document: dict[str, Any] = { + "$schema": SCHEMA_IDS["run"], + "schema_version": 1, + "manifest_type": "run", + "created_at": utc_now(), + "collector": {"name": "qwen-r9700-lab", "version": __version__}, + "inputs": { + "environment": { + "path": str(environment_manifest.expanduser().resolve()), + "file_sha256": sha256_file(environment_manifest), + "manifest_id": environment["manifest_id"], + }, + "model": { + "path": str(model_manifest.expanduser().resolve()), + "file_sha256": sha256_file(model_manifest), + "manifest_id": model["manifest_id"], + }, + }, + "configuration": { + "engine": { + "path": str(engine_config.expanduser().resolve()), + "file_sha256": sha256_file(engine_config), + "config": engine, + }, + "profile": { + "path": str(profile_config.expanduser().resolve()), + "file_sha256": sha256_file(profile_config), + "config": profile, + }, + "reasoning": { + "effort": reasoning_effort, + "enable_thinking": thinking_enabled, + "preserve_thinking": preserve_thinking if thinking_enabled else False, + }, + }, + "execution": {"state": "prepared", "backend_invoked": False}, + } + return write_immutable_manifest(output, document, "run-manifest.schema.json", root) diff --git a/benchmarks/conformance/src/qwen_r9700_lab/ordered_reference_linear.c b/benchmarks/conformance/src/qwen_r9700_lab/ordered_reference_linear.c new file mode 100644 index 0000000..10bb0c6 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/ordered_reference_linear.c @@ -0,0 +1,27 @@ +#include +#include +#include +#include + +_Static_assert(sizeof(float) == 4 && FLT_RADIX == 2 && FLT_MANT_DIG == 24, + "Requires IEEE binary32"); + +/* Each output follows the Python reference's increasing-k multiply/add order. + * No BLAS, fused multiply-add, reassociation or cross-output reduction. + * The caller owns disjoint, contiguous arrays and validates all lengths. + */ +int qwen_ordered_linear(const float *x, const float *w, float *out, + size_t batches, size_t rows, size_t width) { + if (fegetround() != FE_TONEAREST) return 1; + for (size_t batch = 0; batch < batches; ++batch) { + for (size_t row = 0; row < rows; ++row) { + float sum = 0.0f; + for (size_t k = 0; k < width; ++k) { + float product = x[batch * width + k] * w[row * width + k]; + sum = sum + product; + } + out[batch * rows + row] = sum; + } + } + return 0; +} diff --git a/benchmarks/conformance/src/qwen_r9700_lab/ordered_reference_linear.py b/benchmarks/conformance/src/qwen_r9700_lab/ordered_reference_linear.py new file mode 100644 index 0000000..2098d02 --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/ordered_reference_linear.py @@ -0,0 +1,139 @@ +"""Optional CPU implementation of the canonical ordered linear operation. + +This module is not enabled by importing it. A caller must build and bind the +artifact, then explicitly supply its callable to the reference. Compiler and +hardware semantics remain trusted assumptions, not a formal certificate. +""" + +from __future__ import annotations + +import ctypes +import hashlib +import math +import shutil +import subprocess +from pathlib import Path + +import numpy as np + +from qwen_r9700_lab.conformance_artifacts import file_identity +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, authenticate, seal, write_private + +SOURCE = Path(__file__).with_suffix(".c") +FLAGS = ( + "-std=c11", + "-O3", + "-shared", + "-fPIC", + "-fno-fast-math", + "-ffp-contract=off", + "-frounding-math", + "-fno-finite-math-only", +) +SCHEMA = "urn:qwen:ordered-reference-linear:v1" +ARITHMETIC = "increasing-k binary32 multiply then binary32 add; no FMA or reassociation" + + +def build(root: Path) -> dict: + """Compile in a new private artifact directory; never import a GPU library.""" + compiler_name = shutil.which("cc") + if not compiler_name: + raise DiagnosticError("ordered reference linear requires a C compiler") + compiler = Path(compiler_name).resolve() + root = root.resolve() + root.mkdir(mode=0o700) + source = root / "ordered-linear.c" + source.write_bytes(SOURCE.read_bytes()) + source.chmod(0o600) + library = root / "ordered-linear.so" + identities = {"source": file_identity(source), "compiler": file_identity(compiler)} + command = [str(compiler), *FLAGS, str(source), "-lm", "-o", str(library)] + result = subprocess.run(command, capture_output=True, text=True, timeout=120) + log = root / "compiler.log" + log.write_text(result.stdout + result.stderr) + log.chmod(0o600) + if result.returncode: + raise DiagnosticError("ordered reference linear compilation failed; compiler log retained") + if ( + file_identity(source) != identities["source"] + or file_identity(compiler) != identities["compiler"] + ): + raise DiagnosticError("ordered reference compiler or source changed during compilation") + library.chmod(0o600) + binding = seal( + { + "schema": SCHEMA, + "arithmetic": ARITHMETIC, + "flags": list(FLAGS), + "source": {"path": str(source), **identities["source"]}, + "compiler": {"path": str(compiler), **identities["compiler"]}, + "library": {"path": str(library), **file_identity(library)}, + "adapter_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(), + "compiler_and_hardware": "ASSUMED", + "qualification": "BUILT_UNTESTED", + } + ) + write_private(root / "binding.json", binding) + return binding + + +def validate_binding(binding: dict) -> None: + """Validate exact bytes before loading executable code.""" + authenticate(binding) + if ( + binding.get("schema") != SCHEMA + or binding.get("arithmetic") != ARITHMETIC + or binding.get("flags") != list(FLAGS) + or binding.get("adapter_sha256") != hashlib.sha256(Path(__file__).read_bytes()).hexdigest() + or binding.get("source", {}).get("sha256") != file_identity(SOURCE)["sha256"] + ): + raise DiagnosticError("ordered reference implementation binding differs") + for key in ("source", "compiler", "library"): + row = binding[key] + if file_identity(Path(row["path"])) != {"sha256": row["sha256"], "bytes": row["bytes"]}: + raise DiagnosticError(f"ordered reference {key} artifact changed") + + +class OrderedLinear: + """A callable replacement; it never modifies the reference module globally.""" + + def __init__(self, reference, binding: dict): + validate_binding(binding) + self.reference = reference + self.original = reference.linear + self.binding = binding + self.library = ctypes.CDLL(binding["library"]["path"]) + # Check the artifact again after loading; the generated library has no + # user initializers. Artifact immutability during execution is assumed. + validate_binding(binding) + pointer = np.ctypeslib.ndpointer(dtype=np.float32, flags="C_CONTIGUOUS") + self.function = self.library.qwen_ordered_linear + self.function.argtypes = [pointer, pointer, pointer, *([ctypes.c_size_t] * 3)] + self.function.restype = ctypes.c_int + self.calls = 0 + self.fallbacks = 0 + + def __call__(self, value, weight, *, quantize_activation=False, output_bf16=True): + def fallback(): + self.fallbacks += 1 + return self.original( + value, weight, quantize_activation=quantize_activation, output_bf16=output_bf16 + ) + + x, w = np.asarray(value, np.float32), np.asarray(weight, np.float32) + if x.ndim < 1 or w.ndim != 2 or x.shape[-1] != w.shape[-1]: + return fallback() + if quantize_activation: + code, scale = self.reference.activation_quantize(x) + x = np.multiply(self.reference.fp8_decode(code), scale, dtype=np.float32) + x, w = np.ascontiguousarray(x), np.ascontiguousarray(w) + if not np.isfinite(x).all() or not np.isfinite(w).all(): + return fallback() + output = np.empty((*x.shape[:-1], w.shape[0]), np.float32) + status = self.function(x, w, output, math.prod(x.shape[:-1]), w.shape[0], w.shape[-1]) + self.calls += 1 + if status: + raise DiagnosticError("ordered reference linear requires round-to-nearest arithmetic") + if not np.isfinite(output).all(): + return fallback() + return self.reference.bf16(output) if output_bf16 else output diff --git a/benchmarks/conformance/src/qwen_r9700_lab/radiance_cache.py b/benchmarks/conformance/src/qwen_r9700_lab/radiance_cache.py new file mode 100644 index 0000000..21af9ad --- /dev/null +++ b/benchmarks/conformance/src/qwen_r9700_lab/radiance_cache.py @@ -0,0 +1,720 @@ +"""Chat-owned, lossless Radiance snapshots. Also runs over SSH using only Python. + +The model's cache salt isolates chats and compaction generations. The lock lives +outside the generations: retiring one waits for its readers/writers, and late +writes cannot recreate it. Legacy shared caches are never adopted or deleted. +""" + +from __future__ import annotations + +import argparse +import contextlib +import fcntl +import hashlib +import json +import os +import re +import shlex +import shutil +import struct +import subprocess +import tempfile +import threading +import time +import uuid +from datetime import UTC, datetime +from pathlib import Path + +DEFAULT_ROOT = "/home/lewis/.cache/qwen-radiance-public-clean-snapshot-v1" +FORMAT = "qwen-chat-cache-v1" +HEADER = struct.Struct(">8sQ32s") +COMPRESSED = b"QWENKV1Z" +RAW = b"QWENKV1R" +ID = re.compile(r"[0-9a-f]{64}\Z") +KEY = re.compile(r"g[0-9]+-[0-9a-f]{16,128}\.qkv\Z") +CONTROL_DIRECTORY = Path("/dev/shm/qwen-radiance-snapshot-control-v1") +CONTROL_SCHEMA = "urn:qwen-r9700:radiance-snapshot-control:v1" +CONTROL_FILE = re.compile(r"([0-9a-f]{32})\.(request|response)\.json\Z") +_WRITE_LOCKS = tuple(threading.Lock() for _ in range(64)) + + +class RetiredGenerationError(ValueError): + """A valid chat identity names a generation durably superseded by compaction.""" + + +class SnapshotIntegrityError(ValueError): + """Stored bytes fail their own encoding or checksum, independently of the ABI.""" + + +def identity(value: dict) -> dict: + if not isinstance(value, dict) or any( + not isinstance(value.get(k), str) or not ID.fullmatch(value[k]) + for k in ("id", "generation") + ): + raise ValueError("chat id and generation must be SHA256 identifiers") + return { + k: str(value.get(k, ""))[:4096] + for k in ("id", "generation", "title", "cwd", "session_file") + } + + +def cache_salt(chat: dict) -> str: + chat = identity(chat) + return f"{FORMAT}:{chat['id']}:{chat['generation']}" + + +def real_directory(path: Path) -> None: + if path.is_symlink() or not path.is_dir(): + raise ValueError(f"not a real directory: {path}") + + +def sync_directory(path: Path) -> None: + fd = os.open(path, os.O_RDONLY | os.O_DIRECTORY) + try: + os.fsync(fd) + finally: + os.close(fd) + + +def atomic_write(path: Path, content: bytes) -> None: + fd, name = tempfile.mkstemp(prefix=".pending-", dir=path.parent) + temporary = Path(name) + try: + with os.fdopen(fd, "wb") as stream: + stream.write(content) + stream.flush() + os.fsync(stream.fileno()) + temporary.replace(path) + sync_directory(path.parent) + finally: + temporary.unlink(missing_ok=True) + + +def private_control_directory(path: Path = CONTROL_DIRECTORY, *, create: bool = False) -> Path: + """Return the owner-only tmpfs control directory used by the live backend.""" + if create: + path.mkdir(mode=0o700, parents=True, exist_ok=True) + info = path.lstat() + if path.is_symlink() or not path.is_dir() or info.st_uid != os.getuid() or info.st_mode & 0o077: + raise ValueError("snapshot control directory must be private and owner-owned") + return path + + +def request_tail_flush( + chat: dict, + *, + timeout: float = 120.0, + control_directory: Path = CONTROL_DIRECTORY, +) -> dict: + """Ask the live backend to make one in-RAM tail durably publishable.""" + chat = identity(chat) + if not 0 < timeout <= 600: + raise ValueError("snapshot flush timeout must be between 0 and 600 seconds") + directory = private_control_directory(Path(control_directory)) + nonce = uuid.uuid4().hex + request = directory / f"{nonce}.request.json" + response = directory / f"{nonce}.response.json" + payload = { + "schema": CONTROL_SCHEMA, + "nonce": nonce, + "action": "flush", + "chat": chat, + } + atomic_write(request, json.dumps(payload, sort_keys=True).encode()) + deadline = time.monotonic() + timeout + try: + while time.monotonic() < deadline: + try: + info = response.lstat() + if response.is_symlink() or not response.is_file() or info.st_size > 65536: + raise ValueError("unsafe snapshot flush response") + result = json.loads(response.read_text()) + if result.get("schema") != CONTROL_SCHEMA or result.get("nonce") != nonce: + raise ValueError("snapshot flush response identity mismatch") + if result.get("status") == "error": + raise RuntimeError(result.get("error") or "backend snapshot flush failed") + if result.get("status") not in ("flushed", "already_durable"): + raise RuntimeError("backend rejected the snapshot tail flush") + return result + except FileNotFoundError: + time.sleep(0.05) + raise TimeoutError("live backend did not complete the snapshot tail flush") + finally: + request.unlink(missing_ok=True) + response.unlink(missing_ok=True) + + +def encode_block(data: memoryview | bytes) -> bytes: + import zstandard + + compressed = zstandard.ZstdCompressor(level=1, write_checksum=True).compress(data) + magic, payload = (COMPRESSED, compressed) if len(compressed) < len(data) else (RAW, data) + return HEADER.pack(magic, len(data), hashlib.sha256(data).digest()) + bytes(payload) + + +def decode_block(data: bytes, expected_size: int) -> bytes: + import zstandard + + if len(data) < HEADER.size: + raise SnapshotIntegrityError("truncated snapshot header") + magic, size, digest = HEADER.unpack(data[: HEADER.size]) + if size != expected_size: + raise ValueError("snapshot block size differs from the runtime") + payload = data[HEADER.size :] + if magic == COMPRESSED: + # Check the frame size before allocating, even if its header is corrupt. + try: + if zstandard.frame_content_size(payload) != expected_size: + raise SnapshotIntegrityError("snapshot frame size differs from its header") + payload = zstandard.ZstdDecompressor().decompress( + payload, max_output_size=expected_size, allow_extra_data=False + ) + except zstandard.ZstdError as error: + raise SnapshotIntegrityError("invalid compressed snapshot payload") from error + elif magic != RAW: + raise SnapshotIntegrityError("unknown snapshot encoding") + if len(payload) != expected_size or hashlib.sha256(payload).digest() != digest: + raise SnapshotIntegrityError("snapshot checksum mismatch") + return payload + + +class ChatStore: + def __init__(self, data_root: Path | str, chat: dict): + self.chat = identity(chat) + self.root = Path(data_root) + real_directory(self.root) + self.managed = self.root / FORMAT + self.managed.mkdir(exist_ok=True, mode=0o700) + real_directory(self.managed) + self.directory = self.managed / self.chat["id"] + self.directory.mkdir(exist_ok=True, mode=0o700) + real_directory(self.directory) + self.generations = self.directory / "generations" + self.generations.mkdir(exist_ok=True, mode=0o700) + real_directory(self.generations) + self.generation = self.generations / self.chat["generation"] + self._verified = {} + + @contextlib.contextmanager + def lock(self, *, exclusive: bool = False): + fd = os.open(self.directory / ".lock", os.O_CREAT | os.O_RDWR | os.O_NOFOLLOW, 0o600) + try: + fcntl.flock(fd, fcntl.LOCK_EX if exclusive else fcntl.LOCK_SH) + yield + finally: + os.close(fd) + + def metadata(self) -> dict: + path = self.directory / "chat.json" + if path.is_symlink(): + raise ValueError("chat metadata is a symlink") + return json.loads(path.read_text()) if path.exists() else {} + + def save_metadata(self, value: dict) -> None: + value = {**value, "updated_at": datetime.now(UTC).isoformat()} + atomic_write(self.directory / "chat.json", json.dumps(value, sort_keys=True).encode()) + + def current(self) -> bool: + return self.metadata().get("generation") == self.chat["generation"] + + def activate(self) -> dict: + """Retire old writers, retaining one complete fallback until publication.""" + with self.lock(exclusive=True): + prior = self.metadata() + retired = prior.get("retired_generations", []) + if self.chat["generation"] in retired: + raise RetiredGenerationError( + "stale request targets a retired compaction generation" + ) + recreated = not self.generation.exists() + self.generation.mkdir(exist_ok=True, mode=0o700) + real_directory(self.generation) + # Persist the new directory before the tombstone can retire its + # predecessor. Retrying an interrupted activation also repairs it. + sync_directory(self.generations) + if prior.get("generation") != self.chat["generation"]: + if prior.get("generation"): + retired = [*retired, prior["generation"]] + fallback = prior.get("fallback") + if prior.get("head"): + fallback = {k: prior[k] for k in ("generation", "head", "tokens")} + # The durable tombstone precedes deletion. Retrying after a crash + # finishes collection; old queued writers see current() == False. + self.save_metadata( + { + **self.chat, + "format": FORMAT, + "retired_generations": retired, + "status": "empty", + "tokens": 0, + "head": [], + "fallback": fallback, + } + ) + else: + if recreated: + prior = {**prior, "status": "empty", "tokens": 0, "head": []} + self.save_metadata({**prior, **self.chat}) + # An exclusive lock proves no live writer owns these temporary files. + for temporary in self.generation.glob(".pending-*"): + temporary.unlink() + info = self.metadata() + fallback = info.get("fallback") or {} + removed = retained = 0 + for path in self.generations.iterdir(): + if path.name == self.chat["generation"]: + continue + if not ID.fullmatch(path.name): + raise ValueError(f"unexpected generation path: {path}") + real_directory(path) + if path.name == fallback.get("generation"): + keep = set(fallback["head"]) + for block in path.iterdir(): + if block.name in keep: + retained += block.lstat().st_size + elif KEY.fullmatch(block.name) or block.name.startswith(".pending-"): + removed += block.lstat().st_size + block.unlink() + sync_directory(path) + continue + removed += sum(p.stat().st_size for p in path.iterdir() if p.is_file()) + shutil.rmtree(path) + sync_directory(self.generations) + return { + "chat_id": self.chat["id"], + "removed_file_bytes": removed, + "retained_file_bytes": retained, + "generation": self.chat["generation"], + } + + def path(self, key: str) -> Path: + if not KEY.fullmatch(key): + raise ValueError("invalid snapshot object key") + real_directory(self.generation) + path = self.generation / key + if path.is_symlink(): + raise ValueError("snapshot object is a symlink") + return path + + def exists(self, key: str) -> bool: + return self.exists_many([key])[0] + + def exists_many(self, keys: list[str]) -> list[bool]: + # A hybrid prefix lookup can inspect hundreds of keys. Read the chat + # generation once for the batch, rather than reopening its manifest for + # each key on the scheduler thread. + with self.lock(): + if not self.current(): + return [False] * len(keys) + return [self.path(key).is_file() for key in keys] + + def io_totals(self) -> dict: + path = self.directory / "io.json" + if path.is_symlink(): + raise ValueError("snapshot I/O counters are a symlink") + return json.loads(path.read_text()) if path.exists() else {"available": False} + + def _record_io(self, changes: dict) -> None: + """Persist one counter update per transfer batch, outside generations. + + These are completed snapshot payload writes, including objects later + collected. They exclude metadata and failed partial writes; device + counters account for those too. A crash can lose the unfinished batch's + counters, so this is a lower bound rather than a NAND wear estimate. + """ + if not any(changes.values()): + return + fd = os.open(self.directory / ".io.lock", os.O_CREAT | os.O_RDWR | os.O_NOFOLLOW, 0o600) + try: + fcntl.flock(fd, fcntl.LOCK_EX) + prior = self.io_totals() + now = datetime.now(UTC).isoformat() + info = { + **prior, + "available": True, + "since": prior.get("since", now), + "updated_at": now, + "scope": "completed_snapshot_payload_io_lower_bound", + } + for name, amount in changes.items(): + info[name] = prior.get(name, 0) + amount + atomic_write(self.directory / "io.json", json.dumps(info, sort_keys=True).encode()) + finally: + os.close(fd) + + def write(self, key: str, data: memoryview) -> bool: + return self.write_many([(key, data)]) + + def write_many(self, blocks) -> bool: + # Compression holds the read lock too: compaction cannot return before + # all old writes have stopped, and a late writer cannot recreate a folder. + with self.lock(): + if not self.current(): + return False + counts = { + "written_file_bytes": 0, + "written_raw_bytes": 0, + "written_blocks": 0, + "reused_blocks": 0, + "compression_seconds": 0.0, + "write_failures": 0, + } + try: + for key, data in blocks: + path = self.path(key) + # The engine has one process but several filesystem workers. + # Coalesce simultaneous requests for the same immutable key. + with _WRITE_LOCKS[hash(str(path)) % len(_WRITE_LOCKS)]: + if path.exists(): + counts["reused_blocks"] += 1 + continue + started = time.monotonic() + encoded = encode_block(data) + counts["compression_seconds"] += time.monotonic() - started + atomic_write(path, encoded) + counts["written_file_bytes"] += len(encoded) + counts["written_raw_bytes"] += len(data) + counts["written_blocks"] += 1 + except Exception: + counts["write_failures"] += 1 + raise + finally: + self._record_io(counts) + return True + + def read(self, key: str, expected_size: int) -> bytes: + with contextlib.closing(self.read_many([key], expected_size)) as blocks: + return next(blocks) + + def read_many(self, keys: list[str], expected_size: int): + with self.lock(): + if not self.current(): + raise ValueError("snapshot generation was retired") + counts = { + "read_file_bytes": 0, + "read_raw_bytes": 0, + "read_blocks": 0, + "read_failures": 0, + "invalidated_blocks": 0, + "invalidated_file_bytes": 0, + } + try: + for key in keys: + path = self.path(key) + # Exclude a simultaneous repair of this immutable key. A + # reader must not remove a newer valid replacement after + # detecting damage in the old file. Release before yield. + with _WRITE_LOCKS[hash(str(path)) % len(_WRITE_LOCKS)]: + encoded = path.read_bytes() + try: + data = decode_block(encoded, expected_size) + except SnapshotIntegrityError: + # A confirmed corrupt object is not a usable disk + # head. Make it a miss so recomputed data can replace + # it; preserve valid files on ABI or I/O errors. + path.unlink() + sync_directory(path.parent) + counts["invalidated_blocks"] += 1 + counts["invalidated_file_bytes"] += len(encoded) + raise + counts["read_file_bytes"] += len(encoded) + counts["read_raw_bytes"] += len(data) + counts["read_blocks"] += 1 + yield data + except Exception: + counts["read_failures"] += 1 + raise + finally: + self._record_io(counts) + + @staticmethod + def _fingerprint(path: Path) -> list[int]: + value = path.stat() + return [value.st_ino, value.st_size, value.st_mtime_ns, value.st_ctime_ns] + + def prepare_publication(self, keys: list[str], block_size: int) -> dict: + """Verify new/changed payloads away from the scheduler's critical path. + + Published immutable files retain their verified fingerprints. Normal + continuations verify only newly written blocks, not the whole prefix. + """ + import zstandard + + with self.lock(): + prior = self.metadata() + known = ( + prior.get("verified_head", {}) + if prior.get("verified_block_size") == block_size + else {} + ) + result = {"keys": sorted(set(keys)), "missing": [], "invalid": [], "verified": {}} + if prior.get("generation") != self.chat["generation"]: + result["missing"] = result["keys"] + return result + verified_bytes = 0 + for key in result["keys"]: + path = self.path(key) + try: + before = self._fingerprint(path) + if known.get(key) != before and self._verified.get((key, block_size)) != before: + encoded = path.read_bytes() + decode_block(encoded, block_size) + verified_bytes += len(encoded) + if before != self._fingerprint(path): + raise ValueError("snapshot changed during verification") + result["verified"][key] = before + self._verified[key, block_size] = before + except FileNotFoundError: + result["missing"].append(key) + except (OSError, ValueError, zstandard.ZstdError): + result["invalid"].append(key) + self._record_io({"verification_file_bytes": verified_bytes}) + return result + + def publish(self, keys: list[str], tokens: int, block_size: int, *, prepared=None) -> bool: + """Commit a complete head or roll back its abandoned writes, then collect. + + The caller must first drain every request and disk job for this chat. + A rejected successor cannot be repaired after those jobs have finished; + retain the previous head and discard the failed candidate instead. + """ + if prepared is None: + with self.lock(): + if not self.current(): + return False + prepared = self.prepare_publication(keys, block_size) + with self.lock(exclusive=True): + if not self.current(): + return False + prior = self.metadata() + keys = sorted(set(keys)) + if keys != prepared["keys"]: + raise ValueError("verified snapshot differs from publication candidate") + missing, invalid = list(prepared["missing"]), list(prepared["invalid"]) + for key, expected in prepared["verified"].items(): + try: + if self._fingerprint(self.path(key)) != expected: + invalid.append(key) + except FileNotFoundError: + missing.append(key) + except OSError: + invalid.append(key) + complete = not missing and not invalid + info = { + **prior, + "status": "ready" if complete else "incomplete", + "publication": { + "tokens": tokens, + "expected_blocks": len(keys), + "missing_keys": missing, + "invalid_keys": invalid, + "result": "committed" if complete else "rolled_back", + }, + } + if complete: + info.update( + head=keys, + tokens=tokens, + verified_head=prepared["verified"], + verified_block_size=block_size, + fallback=None, + ) + self._verified = { + (key, block_size): value for key, value in prepared["verified"].items() + } + self._collect(info) + return complete + + def _collect(self, info: dict) -> dict: + """Under the exclusive chat lock, persist intent before deleting anything.""" + keep = set(info["head"]) + if any(not KEY.fullmatch(key) for key in keep): + raise ValueError("invalid published head manifest") + info = {**info, "gc": {"status": "pending"}} + self.save_metadata(info) + removed_files = removed_bytes = 0 + try: + for path in self.generation.iterdir(): + if path.name not in keep and ( + KEY.fullmatch(path.name) or path.name.startswith(".pending-") + ): + size = path.lstat().st_size + path.unlink() + removed_files += 1 + removed_bytes += size + fallback_generation = (info.get("fallback") or {}).get("generation") + for directory in self.generations.iterdir(): + if directory.name in {self.chat["generation"], fallback_generation}: + continue + if not ID.fullmatch(directory.name): + raise ValueError("invalid snapshot generation directory") + real_directory(directory) + for path in directory.iterdir(): + removed_files += 1 + removed_bytes += path.lstat().st_size + shutil.rmtree(directory) + sync_directory(self.generation) + sync_directory(self.generations) + except OSError as error: + # If even this write fails, the durable pending record still causes + # startup recovery. Never report a completed GC before directory fsync. + self.save_metadata({**info, "gc": {"status": "failed", "errno": error.errno}}) + raise + result = { + "status": "complete", + "removed_files": removed_files, + "removed_file_bytes": removed_bytes, + } + self.save_metadata({**info, "gc": result}) + return result + + def collect(self, *, failure: str | None = None) -> dict: + """Recover abandoned writes with no active requests, preserving the head.""" + with self.lock(exclusive=True): + if not self.current(): + return {"status": "retired"} + info = self.metadata() + if failure is not None: + info = {**info, "status": "incomplete", "publication": {"result": failure}} + return self._collect(info) + + +def report(data_root: Path) -> dict: + chats = [] + managed = data_root / FORMAT + if managed.exists(): + real_directory(managed) + for directory in sorted(managed.iterdir()): + if not ID.fullmatch(directory.name): + continue + real_directory(directory) + metadata_path = directory / "chat.json" + if not metadata_path.exists(): + continue + info = json.loads(metadata_path.read_text()) + if info.get("id") != directory.name: + raise ValueError("chat metadata identity differs from its directory") + store = ChatStore(data_root, info) + with store.lock(): + info = store.metadata() + sizes = {"files": 0, "file_bytes": 0, "allocated_bytes": 0, "raw_bytes": 0} + for path in store.generations.rglob("*.qkv"): + if path.is_symlink(): + raise ValueError(f"snapshot object is a symlink: {path}") + stat = path.stat() + with path.open("rb") as stream: + header = stream.read(HEADER.size) + sizes["files"] += 1 + sizes["file_bytes"] += stat.st_size + sizes["allocated_bytes"] += stat.st_blocks * 512 + if len(header) == HEADER.size: + sizes["raw_bytes"] += HEADER.unpack(header)[1] + chats.append( + { + **{ + k: v + for k, v in info.items() + if k not in ("head", "retired_generations") + }, + **sizes, + } + ) + legacy_files = legacy_bytes = 0 + for path in data_root.rglob("*.bin"): + if path.is_symlink(): + continue + legacy_files += 1 + legacy_bytes += path.stat().st_size + return { + "data_root": str(data_root), + "chats": chats, + "legacy_unassigned_files": legacy_files, + "legacy_unassigned_bytes": legacy_bytes, + } + + +def human(size: int) -> str: + return f"{size / 1024**3:.2f} GiB" + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--host", default="ai") + parser.add_argument("--cache-root", default=DEFAULT_ROOT) + parser.add_argument("--abi", help="snapshot ABI directory; required for mutation commands") + sub = parser.add_subparsers(dest="command", required=True) + listing = sub.add_parser( + "list", help="list chats, compressed sizes, and unassigned legacy cache" + ) + listing.add_argument("--json", action="store_true") + compact = sub.add_parser("compact", help="retire a chat generation after Pi commits compaction") + compact.add_argument("--identity-json", required=True) + flush = sub.add_parser("flush", help="force the live backend to publish a chat's buffered tail") + flush.add_argument("--identity-json", required=True) + flush.add_argument("--timeout", type=float, default=120.0) + args = parser.parse_args(argv) + if args.host not in ("local", "localhost", "127.0.0.1"): + remote_args = ["--host", "local", "--cache-root", args.cache_root] + if args.abi: + remote_args += ["--abi", args.abi] + remote_args += [args.command] + remote_args += ["--json"] if args.command == "list" and args.json else [] + if args.command in ("compact", "flush"): + remote_args += ["--identity-json", args.identity_json] + if args.command == "flush": + remote_args += ["--timeout", str(args.timeout)] + return subprocess.run( + [ + "ssh", + "-T", + "-o", + "BatchMode=yes", + "-o", + "ConnectTimeout=10", + "--", + args.host, + shlex.join(["python3", "-", *remote_args]), + ], + input=Path(__file__).read_text(), + text=True, + check=False, + ).returncode + snapshots = Path(args.cache_root).expanduser() / "snapshots" + if args.abi and not ID.fullmatch(args.abi): + parser.error("--abi must be a SHA256 identifier") + if args.command in ("compact", "flush") and not args.abi: + parser.error(f"{args.command} requires --abi") + if args.command == "flush": + print(json.dumps(request_tail_flush(json.loads(args.identity_json), timeout=args.timeout))) + return 0 + if args.command == "compact": + result = ChatStore(snapshots / args.abi / "data", json.loads(args.identity_json)).activate() + print(json.dumps(result)) + return 0 + roots = [snapshots / args.abi] if args.abi else sorted(snapshots.iterdir()) + reports = [report(root / "data") for root in roots if (root / "data").is_dir()] + if args.json: + print(json.dumps(reports, indent=2)) + return 0 + for value in reports: + print(value["data_root"]) + print( + f"{'CHAT':12} {'DISK FILES':>12} {'RAW CACHE':>12} " + f"{'TOKENS':>8} {'STATE':10} TITLE / DIRECTORY" + ) + for chat in value["chats"]: + print( + f"{chat['id'][:12]} {human(chat['file_bytes']):>12} " + f"{human(chat['raw_bytes']):>12} " + f"{chat.get('tokens', 0):>8} {chat.get('status', 'unknown'):10} " + f"{chat.get('title') or chat.get('cwd')}" + ) + print( + f"Unassigned legacy cache: {human(value['legacy_unassigned_bytes'])} " + f"({value['legacy_unassigned_files']} files; preserved)" + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/conformance/tests/conformance_fixture.py b/benchmarks/conformance/tests/conformance_fixture.py new file mode 100644 index 0000000..cc330ac --- /dev/null +++ b/benchmarks/conformance/tests/conformance_fixture.py @@ -0,0 +1,103 @@ +"""Public, tiny hybrid checkpoint; no production weights or chat fixtures.""" + +import hashlib +import json + +import numpy as np +from safetensors.numpy import save_file + +from qwen_r9700_lab.conformance_reference import bf16, reference_contract +from qwen_r9700_lab.conformance_replay import PLAN_SCHEMA, reference_code_identity +from qwen_r9700_lab.diagnostic_contract import digest, seal + + +def tiny_checkpoint(root): + root.mkdir() + c = { + "model_type": "qwen3_5_text", + "hidden_size": 32, + "intermediate_size": 64, + "vocab_size": 32, + "max_position_embeddings": 256, + "num_hidden_layers": 4, + "layer_types": ["linear_attention"] * 3 + ["full_attention"], + "hidden_act": "silu", + "rms_norm_eps": 1e-6, + "linear_num_key_heads": 2, + "linear_num_value_heads": 4, + "linear_key_head_dim": 8, + "linear_value_head_dim": 8, + "linear_conv_kernel_dim": 4, + "num_attention_heads": 4, + "num_key_value_heads": 2, + "head_dim": 8, + "attn_output_gate": True, + "rope_parameters": { + "rope_type": "default", + "rope_theta": 1e7, + "partial_rotary_factor": 0.5, + }, + } + rng = np.random.default_rng(90214) + prefix, weights = "model.language_model.", {} + + def tensor(name, shape, scale=0.1): + weights[name] = bf16(rng.normal(0, scale, shape).astype(np.float32)) + + def quantized(name, n, k=32): + weights[name + ".weight"] = rng.integers(0, 256, (n, k // 2), dtype=np.uint8) + weights[name + ".weight_scale"] = np.full((n, k // 32), 122, np.uint8) + + tensor(prefix + "embed_tokens.weight", (32, 32)) + tensor("lm_head.weight", (32, 32)) + tensor(prefix + "norm.weight", (32,)) + for layer, kind in enumerate(c["layer_types"]): + base = prefix + f"layers.{layer}." + for name in ("input_layernorm", "post_attention_layernorm"): + tensor(base + name + ".weight", (32,)) + for name, n, k in (("gate_proj", 64, 32), ("up_proj", 64, 32), ("down_proj", 32, 64)): + quantized(base + "mlp." + name, n, k) + if kind == "linear_attention": + base += "linear_attn." + for name, n in ( + ("in_proj_qkv", 64), + ("in_proj_z", 32), + ("in_proj_a", 4), + ("in_proj_b", 4), + ("out_proj", 32), + ): + quantized(base + name, n) + tensor(base + "conv1d.weight", (64, 1, 4)) + tensor(base + "norm.weight", (8,), 1) + tensor(base + "A_log", (4,)) + tensor(base + "dt_bias", (4,)) + else: + base += "self_attn." + for name, n in (("q_proj", 64), ("k_proj", 16), ("v_proj", 16), ("o_proj", 32)): + quantized(base + name, n) + for name in ("q_norm", "k_norm"): + tensor(base + name + ".weight", (8,)) + save_file(weights, root / "model-00001.safetensors") + (root / "config.json").write_text(json.dumps({"text_config": c})) + (root / "model.safetensors.index.json").write_text( + json.dumps({"weight_map": dict.fromkeys(weights, "model-00001.safetensors")}) + ) + return {p.name: hashlib.sha256(p.read_bytes()).hexdigest() for p in root.iterdir()} + + +def tiny_plan(root): + files = tiny_checkpoint(root) + return seal( + { + "schema": PLAN_SCHEMA, + "contract": digest({"fixture": files, "reference": reference_contract()}), + "execution": digest("CPU fixture"), + "adapter": digest(reference_code_identity()), + "checkpoint": str(root), + "checkpoint_files": files, + "kv_scales": {"3": [1.0, 1.0]}, + "prefix": [1, 4, 8], + "forced_tokens": [7, 9, 13, 3], + "reference_arithmetic": reference_contract(), + } + ) diff --git a/benchmarks/conformance/tests/test_conformance_boundaries.py b/benchmarks/conformance/tests/test_conformance_boundaries.py new file mode 100644 index 0000000..71cc560 --- /dev/null +++ b/benchmarks/conformance/tests/test_conformance_boundaries.py @@ -0,0 +1,122 @@ +import numpy as np +import pytest +from conformance_fixture import tiny_plan + +from qwen_r9700_lab.conformance_boundaries import BoundaryRecorder, compare_boundaries +from qwen_r9700_lab.conformance_replay import run_reference +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, digest, private_json + + +@pytest.mark.parametrize( + "positions,stages", + [ + ([], [["input"]]), + ([0, 0], [["input"]]), + ([-1], [["input"]]), + ([True], [["input"]]), + ([0], [["input", "input"]]), + ([0], [["x/../../escape"]]), + ], +) +def test_invalid_or_empty_observation_domains_are_rejected(tmp_path, positions, stages): + with pytest.raises(DiagnosticError): + BoundaryRecorder( + tmp_path / "capture", + contract=digest("c"), + execution=digest("e"), + adapter=digest("a"), + positions=positions, + layers=1, + layer_stages=stages, + input_digests={p: digest(p) for p in positions}, + ) + + +def test_detailed_reference_really_captures_all_gdn_and_attention_boundaries(tmp_path): + plan = tiny_plan(tmp_path / "checkpoint") + run_reference(plan, tmp_path / "reference") + doc = private_json(tmp_path / "reference" / "semantic" / "boundaries.json") + assert doc["positions"] == list(range(6)) # includes early prefill, not just its final token + stages = doc["layer_stages"] + assert {"conv_input_state", "conv_state", "gdn_input_state", "gdn_state"} <= set(stages[0]) + assert {"attention_q_rope", "attention_key_stored", "attention_output"} <= set(stages[-1]) + assert len(doc["frames"]) == len(doc["positions"]) * sum(map(len, stages)) + result = compare_boundaries( + tmp_path / "reference" / "semantic", + tmp_path / "reference" / "semantic", + tmp_path / "comparison", + ) + assert result["equal"] + + +def test_missing_required_low_level_observation_cannot_finish(tmp_path): + recorder = BoundaryRecorder( + tmp_path / "capture", + contract=digest("c"), + execution=digest("e"), + adapter=digest("a"), + positions=[0], + layers=1, + layer_stages=[["gdn_input_state", "gdn_state"]], + input_digests={0: digest("prefix")}, + ) + recorder.record(0, 0, "gdn_state", np.zeros(2)) + with pytest.raises(DiagnosticError, match="incomplete"): + recorder.finish() + + +def test_prefill_and_accepted_row_domain_omits_emitted_but_pending_token(tmp_path): + from qwen_r9700_lab.conformance_replay import observation_domain, scheduled_inputs + from qwen_r9700_lab.diagnostic_contract import seal + + plan = tiny_plan(tmp_path / "checkpoint") + plan["forced_tokens"] = list(range(32)) + list(range(5)) + plan["accepted_widths"] = list(range(8)) + plan = seal({k: v for k, v in plan.items() if k != "sha256"}) + positions, identities = observation_domain(plan) + tokens = plan["prefix"] + plan["forced_tokens"][:-1] + assert positions == list(range(len(tokens))) + assert len(positions) == list(scheduled_inputs(plan))[-1]["consumed"] + assert all(identities[p] == digest(tokens[: p + 1]) for p in positions) + + +def test_native_boundaries_include_early_prefill_but_not_rejected_speculative_suffix(tmp_path): + from types import SimpleNamespace + + from qwen_r9700_lab.conformance_radiance import RadianceProbe + + observed = [] + probe = RadianceProbe.__new__(RadianceProbe) + probe.index = 0 + probe.campaign = SimpleNamespace(expected=[{"consumed": 3}]) + probe.positions = [0, 1, 2, 3, 4] # two tentative suffix rows + probe.boundaries = SimpleNamespace(record=lambda *args: observed.append(args[:3])) + probe.record_boundary(0, "input_norm", np.zeros((5, 2))) + assert observed == [(0, 0, "input_norm"), (1, 0, "input_norm"), (2, 0, "input_norm")] + + +def test_selected_capture_window_is_explicit_and_never_claims_full_prefill_coverage(tmp_path): + from qwen_r9700_lab.conformance_replay import observation_domain, validate_plan + from qwen_r9700_lab.diagnostic_contract import seal + + plan = tiny_plan(tmp_path / "checkpoint") + plan["observation_positions"] = [1, 4] + plan = seal({k: v for k, v in plan.items() if k != "sha256"}) + validate_plan(plan) + positions, digests = observation_domain(plan) + assert positions == [1, 4] and set(digests) == {1, 4} + run_reference(plan, tmp_path / "reference") + report = private_json(tmp_path / "reference" / "boundaries" / "boundaries.json") + assert report["positions"] == [1, 4] + + +@pytest.mark.parametrize("positions", [[], None, [False], [2, 1], [0, 0], [-1], [6]]) +def test_empty_invalid_or_pending_observation_selection_is_rejected(tmp_path, positions): + from qwen_r9700_lab.conformance_replay import validate_plan + from qwen_r9700_lab.diagnostic_contract import seal + + plan = tiny_plan(tmp_path / "checkpoint") + plan["observation_positions"] = positions + plan = seal({k: v for k, v in plan.items() if k != "sha256"}) + with pytest.raises(DiagnosticError, match="observation positions"): + validate_plan(plan) diff --git a/benchmarks/conformance/tests/test_conformance_control.py b/benchmarks/conformance/tests/test_conformance_control.py new file mode 100644 index 0000000..5ac5468 --- /dev/null +++ b/benchmarks/conformance/tests/test_conformance_control.py @@ -0,0 +1,258 @@ +from copy import deepcopy +from fractions import Fraction + +import pytest + +from qwen_r9700_lab.conformance_control import ( + SCHEMA, + ControlLedger, + ControlRecorder, + audit_control, + read_control_spool, +) +from qwen_r9700_lab.conformance_invariants import interval_certificate, rejection_distribution +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, digest, seal + +GEN = digest("generation A") +COMPONENTS = ["kv", "conv", "gdn"] + + +def event(kind, **fields): + return {"event": kind, **fields} + + +def transcript(width=0): + return [ + event("chat", chat="A", generation=GEN, consumed=32, pending=True, components=COMPONENTS), + event("begin", transaction="t1", chat="A", generation=GEN, drafted=7), + event("fence", transaction="t1"), + event( + "validate", + transaction="t1", + output_equal=True, + state_equal=True, + identity_equal=True, + complete=True, + ), + event( + "commit", + transaction="t1", + accepted=width, + emitted=width + 1, + consumed=33 + width, + pending=True, + versions=dict.fromkeys(COMPONENTS, 33 + width), + ), + ] + + +def audit(events): + return audit_control( + seal( + { + "schema": SCHEMA, + "execution": digest("test execution"), + "adapter": digest("test adapter"), + "events": events, + } + ) + ) + + +@pytest.mark.parametrize("width", range(8)) +def test_every_d7_width_has_exact_pending_and_state_version_conservation(width): + result = audit(transcript(width)) + assert result["equal"] + assert result["counts"]["commits"] == 1 + assert result["native_adapter_qualification"] == "UNPROVED" + + +@pytest.mark.parametrize("component", COMPONENTS) +def test_wrong_component_version_is_not_a_numerical_tolerance(component): + events = transcript(3) + events[-1]["versions"][component] += 1 + result = audit(events) + assert not result["equal"] + assert result["first_failure"]["event_index"] == 4 + + +@pytest.mark.parametrize( + "fault", + [ + "missing_fence", + "state", + "output", + "identity", + "coverage", + "cancelled", + "stale_revision", + "stale_generation", + "pending", + ], +) +def test_deliberate_publication_faults_fail_closed(fault): + events = transcript() + if fault == "missing_fence": + del events[2] + elif fault in {"state", "output", "identity", "coverage"}: + key = "complete" if fault == "coverage" else fault + "_equal" + events[3][key] = False + elif fault == "cancelled": + events.insert(4, event("cancel", transaction="t1")) + elif fault == "stale_revision": + second = deepcopy(events[1:]) + for e in second: + e["transaction"] = "t2" + events = events[:2] + second + events[2:] + elif fault == "stale_generation": + events.insert( + 4, event("generation", chat="A", generation=digest("new"), consumed=3, pending=True) + ) + else: + events[-1]["pending"] = False + assert not audit(events)["equal"] + + +def test_empty_truncated_duplicate_and_unknown_events_never_pass(): + with pytest.raises(DiagnosticError, match="empty"): + audit([]) + assert not audit(transcript()[:-1])["equal"] + assert not audit([*transcript(), event("made_up_event")])["equal"] + assert not audit([*transcript(), transcript()[-1]])["equal"] + + +def ledger_with_two_chats(): + ledger = ControlLedger() + ledger.apply(transcript()[0]) + ledger.apply({**transcript()[0], "chat": "B"}) + return ledger + + +def test_shared_prefix_copy_on_write_preserves_other_chat_and_references(): + ledger = ledger_with_two_chats() + ledger.apply(event("allocate", block=917, chat="A", immutable=True)) + ledger.apply(event("share", block=917, chat="B")) + with pytest.raises(DiagnosticError, match="shared"): + ledger.apply(event("write", block=917, chat="A")) + ledger.apply(event("copy_on_write", source=917, replacement=42, chat="A")) + ledger.apply(event("write", block=42, chat="A")) + assert ledger.blocks[917]["owners"] == {"B"} + with pytest.raises(DiagnosticError, match="different"): + ledger.apply(event("write", block=42, chat="B")) + ledger.apply(event("release", block=42, chat="A")) + assert 42 not in ledger.blocks + assert ledger.blocks[917]["immutable"] + + +def test_failed_copy_on_write_is_atomic_and_external_pin_prevents_writes(): + ledger = ledger_with_two_chats() + ledger.apply(event("allocate", block=1, chat="A", immutable=False, external=1)) + before = deepcopy(ledger.blocks) + with pytest.raises(DiagnosticError): + ledger.apply(event("copy_on_write", source=1, replacement=1, chat="A")) + assert ledger.blocks == before + with pytest.raises(DiagnosticError, match="pinned"): + ledger.apply(event("write", block=1, chat="A")) + with pytest.raises(DiagnosticError, match="immutable"): + ledger.apply(event("share", block=1, chat="B")) + + +def test_verified_previous_snapshot_survives_failed_replacement_and_compaction(): + ledger = ledger_with_two_chats() + for name in ("old", "replacement"): + ledger.apply( + event( + "snapshot", + snapshot=name, + chat="A", + generation=GEN, + consumed=32, + state_sha256=digest(name), + ) + ) + ledger.apply(event("verify_snapshot", snapshot="old", state_sha256=digest("old"), durable=True)) + ledger.apply(event("publish_snapshot", snapshot="old")) + with pytest.raises(DiagnosticError): + ledger.apply(event("publish_snapshot", snapshot="replacement")) + assert ledger.chats["A"]["head"] == "old" + ledger.apply( + event( + "restore", + snapshot="old", + chat="A", + generation=GEN, + consumed=32, + state_sha256=digest("old"), + ) + ) + with pytest.raises(DiagnosticError): + ledger.apply( + event( + "restore", + snapshot="old", + chat="B", + generation=GEN, + consumed=32, + state_sha256=digest("old"), + ) + ) + ledger.apply( + event("generation", chat="A", generation=digest("compacted"), consumed=4, pending=True) + ) + ledger.apply( + event( + "verify_snapshot", + snapshot="replacement", + state_sha256=digest("replacement"), + durable=True, + ) + ) + with pytest.raises(DiagnosticError): + ledger.apply(event("publish_snapshot", snapshot="replacement")) + assert ledger.chats["A"]["head"] == "old" + + +def test_content_free_spool_records_real_order_and_detects_lost_or_reordered_events(tmp_path): + recorder = ControlRecorder( + tmp_path / "events", execution=digest("exec"), adapter=digest("adapter") + ) + for e in transcript(): + recorder.record(e["event"], **{k: v for k, v in e.items() if k != "event"}) + assert recorder.finish()["equal"] + first, second = (tmp_path / "events" / f"{n:09d}.json" for n in (0, 1)) + first.write_bytes(second.read_bytes()) + with pytest.raises(DiagnosticError, match="reordered"): + read_control_spool(tmp_path / "events", count=5) + + +def test_interval_certificate_rejects_ties_and_missing_true_winner(): + assert interval_certificate(["2", "0"], ["3", "1"], 0, bound_origin="fixture")[ + "certified_under_bounds" + ] + assert not interval_certificate(["1", "0"], ["2", "1"], 0, bound_origin="fixture")[ + "certified_under_bounds" + ] + result = interval_certificate( + ["2", "0", "4"], ["3", "1", "5"], 0, bound_origin="shortlist omitted token 2" + ) + assert not result["certified_under_bounds"] + assert result["bounds_soundness"] == "ASSUMED" + with pytest.raises(DiagnosticError): + interval_certificate([1, 2], [3], 0, bound_origin="fixture") + + +@pytest.mark.parametrize( + "p,q", + [ + (["1/2", "1/2"], ["1/2", "1/2"]), + (["1", "0"], ["0", "1"]), + (["0", "1/3", "2/3"], ["1/4", "3/4", "0"]), + ], +) +def test_exact_rejection_reference_covers_zero_draft_and_full_acceptance(p, q): + result = rejection_distribution(p, q) + assert result["output"] == [Fraction(v) for v in p] + if result["rejected_mass"]: + assert sum(result["residual"]) == 1 + else: + assert result["residual"] is None diff --git a/benchmarks/conformance/tests/test_conformance_execution_modes.py b/benchmarks/conformance/tests/test_conformance_execution_modes.py new file mode 100644 index 0000000..b000e7e --- /dev/null +++ b/benchmarks/conformance/tests/test_conformance_execution_modes.py @@ -0,0 +1,236 @@ +"""Reject confounded experiments and unbound outputs before reporting agreement.""" + +from copy import deepcopy + +import numpy as np +import pytest + +from qwen_r9700_lab.conformance_execution_modes import ( + SOURCES, + admit_pair, + compare_pair, + numerical_config, +) +from qwen_r9700_lab.conformance_topk import summarize_logits +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, seal + + +def reseal(document): + return seal({k: v for k, v in document.items() if k != "sha256"}) + + +def saved_rows(change=False): + logits = summarize_logits(np.arange(32, dtype=np.float32)) + altered = summarize_logits(-np.arange(32, dtype=np.float32)) + return seal( + { + "schema": "urn:qwen:d7-equivalence-private-rows:v1", + "continuation": "f" * 64, + "prefill": altered if change else logits, + "rows": [ + { + "position": p, + "absolute_position": 60000 + p, + "target_rows": 8, + "logits": altered if change and p == 16 else logits, + } + for p in range(320) + ], + } + ) + + +def side(mode, rows, capture=False): + graph = "NONE" if mode != "compiled" or capture else "PIECEWISE" + return { + "measurement": seal( + { + "execution_mode": mode, + "correctness": True, + "m1": False, + "lanes": ["fixed-bf16"], + "isolated_capture": capture, + "fixture": "f" * 64, + "binding": "b" * 64, + "driver_sha256": "d" * 64, + "prefix_tokens": 60000, + } + ), + "config": seal( + { + "enforce_eager": mode == "eager", + "max_num_seqs": 2, + "max_num_batched_tokens": 2048, + "compilation_config": { + "mode": 0 if mode == "eager" else 3, + "cudagraph_mode": graph, + }, + "worker_cls": "same-integration", + } + ), + "runtime": seal( + { + "enforce_eager": mode == "eager", + "compilation_mode": 0 if mode == "eager" else 3, + "graph_mode": graph, + "async_scheduling": False, + "effective_capacity": { + "max_num_seqs": 2, + "max_num_batched_tokens": 2048, + "max_model_len": 253792, + "block_size": 16, + "cache_dtype": "fp8", + "num_gpu_blocks": 16000, + }, + "diagnostic_sources": dict.fromkeys(SOURCES, "a" * 64), + "repair": {"bundle": "r" * 64}, + "performance": {"manifest": "p" * 64}, + "runtime": seal( + { + "python": "pinned", + "kernel": "pinned", + "machine": "x86_64", + "packages": {"torch": "pinned"}, + "flags": {}, + } + ), + } + ), + "pass": seal( + { + "mode": "correctness", + "execution_mode": mode, + "fixture": "f" * 64, + "observation": { + "observation": { + "counts": { + "target_forward_calls": 41, + "target_graph_replays": 2600 if graph != "NONE" else 0, + } + }, + "forced": {"sha256": rows["sha256"]}, + }, + } + ), + } + + +@pytest.mark.parametrize( + "left_mode,right_mode,capture", + [ + ("compiled", "compiled-no-graphs", False), + ("compiled-no-graphs", "eager", False), + ("compiled-no-graphs", "compiled-no-graphs", True), + ("eager", "eager", True), + ], +) +def test_admits_controlled_modes_and_observer_bridges(left_mode, right_mode, capture): + rows = saved_rows() + report = compare_pair(side(left_mode, rows), side(right_mode, rows, capture), rows, rows) + assert report["decode"]["positions"] == 320 + assert report["decode"]["full_logits_exact"] == 320 + assert report["prefill"]["full_logits_exact"] == 1 + assert report["stage_localization"].startswith("UNMEASURED") + + +def test_keeps_prefill_disagreement_separate_from_decode(): + a, b = saved_rows(), saved_rows(change=True) + result = compare_pair(side("compiled-no-graphs", a), side("eager", b), a, b) + assert result["prefill"]["full_logits_exact"] == 0 + assert result["decode"]["full_logits_exact"] == 319 + assert result["decode"]["1"]["set_exact"] == 319 + + +def test_m1_observer_requires_explicit_arm_and_both_sides_m1(): + rows = saved_rows() + for row in rows["rows"]: + row["target_rows"] = 1 + rows = reseal(rows) + a, b = side("compiled", rows), side("compiled-no-graphs", rows, capture=True) + for record in (a, b): + record["measurement"]["m1"] = True + record["measurement"] = reseal(record["measurement"]) + with pytest.raises(DiagnosticError): + compare_pair(a, b, rows, rows) + result = compare_pair(a, b, rows, rows, arm="m1") + assert result["admission"]["arm"] == "m1" + assert result["decode"]["full_logits_exact"] == 320 + b["measurement"]["m1"] = False + b["measurement"] = reseal(b["measurement"]) + with pytest.raises(DiagnosticError): + compare_pair(a, b, rows, rows, arm="m1") + + +@pytest.mark.parametrize( + "section,field,value", + [ + ("measurement", "binding", "different"), + ("measurement", "driver_sha256", "different"), + ("measurement", "fixture", "different"), + ("measurement", "prefix_tokens", 57008), + ("measurement", "m1", True), + ("config", "max_num_seqs", 1), + ("config", "max_num_batched_tokens", 4096), + ("config", "unrecognized_future_numeric_flag", True), + ("runtime", "repair", {"bundle": "different"}), + ("runtime", "performance", {"manifest": "different"}), + ("runtime", "diagnostic_sources", dict.fromkeys(SOURCES, "different")), + ("runtime", "effective_capacity", {}), + ("runtime", "async_scheduling", True), + ], +) +def test_rejects_confounds_even_when_outputs_happen_to_match(section, field, value): + rows = saved_rows() + a, b = side("compiled-no-graphs", rows), side("eager", rows) + b[section][field] = value + b[section] = reseal(b[section]) + with pytest.raises(DiagnosticError): + compare_pair(a, b, rows, rows) + + +def test_keeps_unknown_compiler_options_in_the_comparison(): + base = {"compilation_config": {"mode": 0, "custom_ops": ["+all"]}} + changed = deepcopy(base) + changed["compilation_config"]["custom_ops"] = ["-all"] + assert numerical_config(base) != numerical_config(changed) + + +def test_rejects_changed_effective_capacity_and_runtime_flags(): + rows = saved_rows() + for variant in ("capacity", "flag"): + a, b = side("compiled-no-graphs", rows), side("eager", rows) + if variant == "capacity": + b["runtime"]["effective_capacity"]["max_num_seqs"] = 1 + else: + b["runtime"]["runtime"]["flags"]["NUMERIC_FLAG"] = "1" + b["runtime"]["runtime"] = reseal(b["runtime"]["runtime"]) + b["runtime"] = reseal(b["runtime"]) + with pytest.raises(DiagnosticError): + admit_pair(a, b) + + +def test_rejects_rows_from_another_pass_and_incomplete_or_wrong_width_replays(): + good, other = saved_rows(), saved_rows(change=True) + a, b = side("compiled-no-graphs", good), side("eager", good) + with pytest.raises(DiagnosticError, match="observed pass"): + compare_pair(a, b, good, other) + for fault in ("short", "width", "position"): + bad = deepcopy(good) + if fault == "short": + bad["rows"].pop() + elif fault == "width": + bad["rows"][0]["target_rows"] = 1 + else: + bad["rows"][0]["absolute_position"] += 1 + bad = reseal(bad) + with pytest.raises(DiagnosticError): + compare_pair(a, side("eager", bad), good, bad) + + +def test_rejects_claimed_graph_mode_without_actual_replay(): + rows = saved_rows() + a, b = side("compiled", rows), side("compiled-no-graphs", rows) + a["pass"]["observation"]["observation"]["counts"]["target_graph_replays"] = 0 + a["pass"] = reseal(a["pass"]) + with pytest.raises(DiagnosticError, match="observed graph execution"): + admit_pair(a, b) diff --git a/benchmarks/conformance/tests/test_conformance_gdn_contract.py b/benchmarks/conformance/tests/test_conformance_gdn_contract.py new file mode 100644 index 0000000..df52788 --- /dev/null +++ b/benchmarks/conformance/tests/test_conformance_gdn_contract.py @@ -0,0 +1,80 @@ +from copy import deepcopy +from pathlib import Path + +import pytest + +from qwen_r9700_lab import conformance_gdn_contract as contract +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, seal + +ROOT = Path(__file__).resolve().parents[1] +PATH = ROOT / "configs/profiles/gdn-stock-m1-arithmetic-v1.json" + + +def test_reference_choice_cannot_be_silently_relabelled(tmp_path): + import json + + original = contract.read_contract(PATH) + altered = deepcopy(original) + altered.pop("sha256") + altered["precision"]["beta"] = "FP32 with no BF16 rounding" + changed = tmp_path / "contract.json" + changed.write_text(json.dumps(seal(altered))) + with pytest.raises(DiagnosticError, match="unreviewed"): + contract.read_contract(changed) + + +def test_a_resealed_contract_does_not_admit_a_different_compiler(tmp_path): + original = contract.read_contract(PATH) + altered = deepcopy(original) + altered.pop("sha256") + altered["arithmetic_authority"]["compiler"]["enable_fp_fusion"] = False + with pytest.raises(DiagnosticError, match="unreviewed"): + contract.audit_artifacts(seal(altered), tmp_path, {}) + + +def test_evidence_audit_requires_both_the_kernel_and_its_math_helpers(tmp_path): + with pytest.raises(DiagnosticError, match="incomplete"): + contract.audit_artifacts(contract.read_contract(PATH), tmp_path, {}) + + +def test_edited_source_cannot_pass_as_the_pinned_stock_reference(tmp_path): + source = tmp_path / "fused_recurrent.py" + source.write_text("# modified beta or normalization arithmetic\n") + with pytest.raises(DiagnosticError, match="evidence changed"): + contract.audit_artifacts( + contract.read_contract(PATH), + tmp_path, + {"fused_recurrent.py": source, "fla_op.py": tmp_path / "not-read.py"}, + ) + + +@pytest.mark.parametrize("accepted", range(8)) +def test_acceptance_counts_processed_pending_input_but_not_new_bonus(accepted): + rows = ("old_pending", *(f"proposal_{i}" for i in range(1, 8))) + processed = tuple(rows[i] for i in contract.required_processed_rows(accepted)) + assert processed == rows[: accepted + 1] + assert rows[contract.committed_gdn_row(accepted)] == processed[-1] + assert len(processed) == accepted + 1 + + +@pytest.mark.parametrize("accepted", [-1, 8, True, 1.0, "1", None]) +def test_invalid_rejection_widths_fail_closed(accepted): + with pytest.raises(DiagnosticError): + contract.required_processed_rows(accepted) + + +def test_different_rounding_and_contraction_are_observable(): + import numpy as np + + from qwen_r9700_lab.conformance_reference import bf16 + + halfway = np.array([0x3F808000], dtype=np.uint32) + assert bf16(halfway.view(np.float32)).view(np.uint32)[0] == 0x3F800000 + assert ((halfway + np.uint32(0x8000)) & np.uint32(0xFFFF0000))[0] == 0x3F810000 + # These particular operands have an exact FP64 product and sum, making this + # a valid finite counterexample, not a general FP64 emulation of FP32 FMA. + a, b = np.float32(1 + 2**-23), np.float32(1 - 2**-23) + separate = np.float32(np.float32(a * b) - np.float32(1)) + fused_for_these_operands = np.float32(np.float64(a) * np.float64(b) - 1) + assert separate == 0 + assert fused_for_these_operands == -(2**-46) diff --git a/benchmarks/conformance/tests/test_conformance_gpu_lease.py b/benchmarks/conformance/tests/test_conformance_gpu_lease.py new file mode 100644 index 0000000..50434dd --- /dev/null +++ b/benchmarks/conformance/tests/test_conformance_gpu_lease.py @@ -0,0 +1,157 @@ +"""Process-level ownership tests; no GPU imports or device access.""" + +import multiprocessing +import os +import sys + +import pytest + +from qwen_r9700_lab.conformance_gpu_lease import block_cleanup, cleanup_block, gpu_lease +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, private_json + + +def hold(path, evidence, acquired, release): + os.environ["QWEN_CONFORMANCE_GPU_LOCK"] = str(path) + with gpu_lease(evidence): + acquired.set() + if not release.wait(15): + raise TimeoutError("test did not release its worker") + + +def die_holding(path, evidence): + os.environ["QWEN_CONFORMANCE_GPU_LOCK"] = str(path) + with gpu_lease(evidence): + os._exit(9) + + +def test_two_controllers_never_hold_gpu_lease_together(tmp_path): + context = multiprocessing.get_context("spawn") + acquired = [context.Event(), context.Event()] + release = [context.Event(), context.Event()] + workers = [ + context.Process( + target=hold, args=(tmp_path / "lock", tmp_path / f"p{i}", acquired[i], release[i]) + ) + for i in range(2) + ] + try: + workers[0].start() + assert acquired[0].wait(10) + workers[1].start() + assert not acquired[1].wait(0.2) + release[0].set() + assert acquired[1].wait(10) + release[1].set() + for worker in workers: + worker.join(10) + assert worker.exitcode == 0 + assert private_json(tmp_path / "lock.owner.json")["status"] == "released" + finally: + for event in release: + event.set() + for worker in workers: + if worker.pid is not None: + if worker.is_alive(): + worker.terminate() + worker.join(10) + + +def test_unclean_owner_blocks_next_case_until_cleanup_is_reviewed(tmp_path, monkeypatch): + context = multiprocessing.get_context("spawn") + path = tmp_path / "lock" + worker = context.Process(target=die_holding, args=(path, tmp_path / "dead")) + worker.start() + worker.join(10) + if worker.is_alive(): + worker.terminate() + worker.join(10) + pytest.fail("fault worker did not exit") + assert worker.exitcode == 9 + monkeypatch.setenv("QWEN_CONFORMANCE_GPU_LOCK", str(path)) + with ( + pytest.raises(DiagnosticError, match="did not release cleanly"), + gpu_lease(tmp_path / "next"), + ): + pytest.fail("an unreviewed dead owner must prevent admission") + + +def test_python_error_releases_after_owned_gpu_scope_unwinds(tmp_path, monkeypatch): + monkeypatch.setenv("QWEN_CONFORMANCE_GPU_LOCK", str(tmp_path / "lock")) + with pytest.raises(ValueError), gpu_lease(tmp_path / "first"): + raise ValueError("synthetic failure") + with gpu_lease(tmp_path / "second"): + assert private_json(tmp_path / "lock.owner.json")["status"] == "active" + + +def test_unconfigured_lease_does_not_create_resources(tmp_path, monkeypatch): + monkeypatch.delenv("QWEN_CONFORMANCE_GPU_LOCK", raising=False) + with gpu_lease(tmp_path / "unused"): + pass + assert not (tmp_path / "unused").exists() + + +def test_symlink_lock_is_rejected(tmp_path, monkeypatch): + target, path = tmp_path / "target", tmp_path / "lock" + target.touch() + path.symlink_to(target) + monkeypatch.setenv("QWEN_CONFORMANCE_GPU_LOCK", str(path)) + with pytest.raises(OSError), gpu_lease(tmp_path / "evidence"): + pytest.fail("symlink lock admitted") + + +def test_surviving_worker_blocks_next_lease_and_repeated_close(tmp_path, monkeypatch): + from qwen_r9700_lab import conformance_transport as transport + + monkeypatch.setenv("QWEN_CONFORMANCE_GPU_LOCK", str(tmp_path / "lock")) + # The real launcher exits; the injected observation models a kernel-stuck + # descendant that cannot be created safely in an ordinary regression test. + monkeypatch.setattr( + transport, "await_group_exit", lambda group: [{"pid": group + 1, "state": "D"}] + ) + with pytest.raises(DiagnosticError, match="remain alive"), gpu_lease(tmp_path / "first"): + child = transport.OwnedProcess( + [sys.executable, "-c", "pass"], tmp_path / "child", env=dict(os.environ), timeout=10 + ) + child.wait() + assert child.process.poll() is not None + assert child.log.closed + assert cleanup_block()["members"][0]["state"] == "D" + assert private_json(tmp_path / "lock.owner.json")["status"] == "cleanup_incomplete" + assert not (tmp_path / "first/released.json").exists() + with pytest.raises(DiagnosticError, match="remain alive") as caught: + child.close() + assert caught.value is child.close_error + with pytest.raises(DiagnosticError, match="admission blocked"), gpu_lease(tmp_path / "next"): + pytest.fail("a dead launcher does not prove its workers released the GPU") + + +def test_diagnostic_write_failure_does_not_certify_clean_release(tmp_path, monkeypatch): + monkeypatch.setenv("QWEN_CONFORMANCE_GPU_LOCK", str(tmp_path / "lock")) + with pytest.raises(FileNotFoundError), gpu_lease(tmp_path / "scope"): + block_cleanup( + tmp_path / "missing-directory", process_group=17, reason="fixture", members=[] + ) + assert cleanup_block() is not None + assert private_json(tmp_path / "lock.owner.json")["status"] == "cleanup_incomplete" + assert not (tmp_path / "scope/released.json").exists() + + +def test_failure_to_write_block_marker_still_preserves_unclean_owner(tmp_path, monkeypatch): + from qwen_r9700_lab import conformance_gpu_lease as lease + + monkeypatch.setenv("QWEN_CONFORMANCE_GPU_LOCK", str(tmp_path / "lock")) + replace = lease.replace_private + + def unavailable(root, name, document): + if name.endswith(".blocked.json"): + raise OSError("synthetic marker write failure") + return replace(root, name, document) + + monkeypatch.setattr(lease, "replace_private", unavailable) + with pytest.raises(OSError, match="marker write failure"), gpu_lease(tmp_path / "scope"): + block_cleanup(tmp_path / "scope", process_group=17, reason="fixture", members=[]) + assert not (tmp_path / "lock.blocked.json").exists() + assert private_json(tmp_path / "lock.owner.json")["status"] == "cleanup_incomplete" + assert not (tmp_path / "scope/released.json").exists() + with pytest.raises(DiagnosticError, match="admission blocked"), gpu_lease(tmp_path / "next"): + pytest.fail("a diagnostic I/O failure must not permit another GPU job") diff --git a/benchmarks/conformance/tests/test_conformance_instrumentation.py b/benchmarks/conformance/tests/test_conformance_instrumentation.py new file mode 100644 index 0000000..2c03814 --- /dev/null +++ b/benchmarks/conformance/tests/test_conformance_instrumentation.py @@ -0,0 +1,224 @@ +import functools +from types import SimpleNamespace + +import numpy as np +import pytest + +from qwen_r9700_lab.conformance_instrumentation import ( + CallRecorder, + HookSet, + callable_identity, + compare_calls, +) +from qwen_r9700_lab.conformance_state import load_arrays +from qwen_r9700_lab.diagnostic_contract import DiagnosticError, digest + + +class Layer: + def __init__(self, fault=False): + self.fault = fault + + def inner(self, x): + x[0] += 1 + int(self.fault) + return x * np.float32(2) + + def forward(self, x): + return self.inner(x) + np.float32(3) + + +class CustomOpDef: + __module__ = "torch._library.custom_ops" + + def __init__(self, fn): + self._qualname = "qualification::example" + self._schema = "(Tensor x) -> Tensor" + self._init_fn = fn + self._backend_fns = {} + self._disabled_kernel = set() + self.register(fn) + + def register(self, fn): + def wrapped_fn(*args, **kwargs): + return fn(*args, **kwargs) + + self._backend_fns["cpu"] = wrapped_fn + + def __call__(self, value): + return self._backend_fns["cpu"](value) + + +def test_custom_operator_binding_tracks_registered_implementation(tmp_path): + def initial(x): + return x + 1 + + def replacement(x): + return x + 2 + + operator = CustomOpDef(initial) + before = callable_identity(operator) + operator.register(replacement) + assert callable_identity(operator) != before + operator.register(initial) + assert callable_identity(operator) == before + owner, hooks = SimpleNamespace(op=operator), HookSet() + rec = recorder(tmp_path / "custom-op", required=("custom",)) + rec.bind(owner, "op", site="custom", hooks=hooks, source_sha256=before) + np.testing.assert_array_equal(owner.op(np.zeros(3)), np.ones(3)) + operator.register(replacement) + with pytest.raises(DiagnosticError, match="registration changed"): + owner.op(np.zeros(3)) + rec.finish() + hooks.close() + assert owner.op is operator + + +def recorder(path, *, mode="tensor", export=np.asarray, required=("layer", "inner")): + return CallRecorder( + path, + contract=digest("contract"), + execution=digest("execution"), + adapter=digest("adapter"), + tensor_export=export, + is_tensor=lambda v: isinstance(v, np.ndarray), + mode=mode, + required_sites=required, + ) + + +def record_model(path, fault=False): + rec, hooks, layer = recorder(path), HookSet(), Layer(fault) + rec.bind(layer, "forward", site="layer", hooks=hooks) + rec.bind(layer, "inner", site="inner", hooks=hooks) + value = np.arange(4, dtype="