diff --git a/.github/workflows/witnesses.yml b/.github/workflows/witnesses.yml index ef9a732b65a..7cf601ba4dc 100644 --- a/.github/workflows/witnesses.yml +++ b/.github/workflows/witnesses.yml @@ -20,6 +20,7 @@ env: GUNBC_REQUIRED_FLOOR_DISPOSITION: required_floor_disposition.tsv GUNBC_LONG_HOME_STORAGE_AGREEMENT: long_home_storage_agreement.tsv GUNBC_REQUIRED_FLOOR_CLAIM_COST: required_floor_claim_cost.tsv + GUNBC_REQUIRED_FLOOR_CROSS_CLAIM_DEMAND: required_floor_cross_claim_demand.tsv GUNBC_REQUIRED_CI_CONTRACT_EPOCH: 2026-09-01.1 jobs: required-witnesses-build: @@ -173,6 +174,14 @@ jobs: if-no-files-found: error retention-days: 14 if: "!cancelled() && steps.build_witness_fold.outcome == 'success'" + - name: Upload the floor's cross-claim demand census + uses: actions/upload-artifact@v4 + with: + name: required-floor-cross-claim-demand + path: required_floor_cross_claim_demand.tsv + if-no-files-found: error + retention-days: 14 + if: "!cancelled() && steps.build_witness_fold.outcome == 'success'" - name: Record runner filesystem at job end run: | # dissolve-on: toolchain_filesystem_probe -- delete the start/end runner-filesystem instrument after its joined readings identify and the fleet fixes the toolchain deleter, OR after per-job runner microVMs make the shared filesystem eviction class impossible diff --git a/DESIGN.md b/DESIGN.md index c2241f4d002..4e844d2e8cd 100644 --- a/DESIGN.md +++ b/DESIGN.md @@ -240,6 +240,7 @@ One row per class, each carrying its recognition rule and its receipts, in [docs - `ambient_process_state_read_by_a_concurrent_reader` - `predicate_vacuously_true_on_an_empty_domain` - `check_subject_narrower_than_its_declared_claim` +- `recurrence_ledger_scoped_below_the_recurrence` ## Building & checks diff --git a/dag/gunbc/floor/cross_claim_demand_census_seed_growth.dag b/dag/gunbc/floor/cross_claim_demand_census_seed_growth.dag new file mode 100644 index 00000000000..561adb66f31 --- /dev/null +++ b/dag/gunbc/floor/cross_claim_demand_census_seed_growth.dag @@ -0,0 +1,43 @@ +module gunbc.cross_claim_demand_census_seed_growth + +import gunbc.roadmap_model { RoadmapNodeId } +import gunbc.seed_growth { SeedGrowthJustification } +import std.decl_ref { DeclarationRef, WholeDeclaration } + +// FORWARD-FREEZE RECEIPT for the cross-claim demand census: the required floor reporting, as a run +// product, which pure producer identities it re-derived across claim frames. +// +// WHAT THE ARTIFACT IS FOR, stated first because a justification that only says what the code does +// cannot be checked. `v2.workflow.floor_pure_producer_share` is the mechanism that stops the floor +// re-deriving a pure producer once per claim, and its roster is HAND-AUTHORED. Nothing in the +// repository produced its candidates, because the interpreter's recompute ledger is scoped to ONE +// evaluation frame and reports only keys re-hit inside it: a producer evaluated exactly once per +// claim, across thousands of claims, carries count=1 in every frame's ledger and appears in none of +// them. The instrument that exists to rank redundant recompute was structurally blind to the whole +// population its own repair roster is enrolled from, so candidates were discovered when a per-claim +// budget refusal landed on an unrelated lane's pull request. This artifact names the next candidate +// instead. +data cross_claim_demand_census_seed_growth_justification: SeedGrowthJustification = SeedGrowthJustification { + hand_authored_declarations: [ + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "CrossClaimDemandRow", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "CrossClaimDemandKey", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "CrossClaimDemandCell", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "CrossClaimDemandCensus", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "CROSS_CLAIM_DEMAND", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "CROSS_CLAIM_DEMAND_RETENTION_FLOOR_NS", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "CROSS_CLAIM_DEMAND_KEY_CAP", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "CROSS_CLAIM_DEMAND_MODULE_SAMPLE_CAP", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "cross_claim_demand_args_hash", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "cross_claim_demand_absorb_one", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "absorb_claim_recompute_demand", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "cross_claim_demand_rows", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "cross_claim_demand_disclosure", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "clear_cross_claim_demand_census", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.v1_interpreter", decl_name: "eval_recompute_decl_site", field: WholeDeclaration }, + DeclarationRef { module_path: "v1_compiler.cli_run", decl_name: "write_required_floor_cross_claim_demand_tsv", field: WholeDeclaration } + ], + reason: "WHY RUST IS STILL NEEDED, and it is the same boundary cross_claim_pure_share_seed_growth names rather than a second one: the floor builds a FRESH EVALUATION FRAME PER CLAIM, deliberately, so one witness cannot contaminate the next -- and the substrate carries no modeled claim-frame boundary, so nothing authored in .dag can observe across it. The recompute ledger this census folds is already interpreter state; the census stands exactly where the frames are built and does nothing else there.\n\nWHAT IS NOT GROWN: no policy, no threshold anyone refuses on, no roster membership, no escape hatch, and no second producer. The census READS the ledger the interpreter already keeps under GUNBC_RECOMPUTE_TRACE and adds one aggregation across the frame boundary; enrolment stays entirely with v2.workflow.floor_pure_producer_share, which decides membership on a criterion this artifact deliberately cannot supply. A row here is a CANDIDATE whose SERVE cost is unmeasured -- that roster's own header records the case where the serve lost to the recompute and two enrolled rows were REMOVED -- so reading a top row as an enrolment instruction would trade a measured red for an unmeasured regression, and the artifact's header says so.\n\nEVERY TRUNCATION IS DISCLOSED, which is the one place this instrument could have failed the way its subject did. Sub-millisecond first sightings are not retained at identity grain and the omitted key count and their summed cost are printed and written; the distinct-key cap counts its refusals; the per-row module sample is bounded while the module COUNT is exact; and single-claim rows are RETAINED rather than filtered, so the shared population has a control beside it. An artifact that truncated silently would be read as a population, which is gunbc.recurring_failure_mode instrument_output_read_as_subject_content -- the class this census would otherwise commit while reporting on its own cause.\n\nWHY IT IS ADMITTED AGAINST THE v1 FREEZE: gunbc.v1_maintenance_standing v1_seed_standing admits work serving the v2 self-host program, and the required floor is the instrument that program gates on. Measured on main run 33615659836 (required_floor_claim_cost.tsv, 3477 executed rows, cost_basis=cpu): p50 is 2ms and 3379 rows sit under 280ms, while a tail family sits against the 500ms per-claim ceiling -- and inside that family a witness and its DISCRIMINATING RED, which do materially different work, cost within 2ms of each other. A cost that does not move when the assertion changes is not the assertion's cost. One 34-row module spans 0ms to the ceiling in three bands that track which shared producer a row forces rather than what it asserts. That is the per-claim re-derivation, and while it is invisible to every instrument this repository runs, the budget refusal it causes lands on whichever unrelated lane is pushing.\n\nTHE COST COLUMN IS NOT ADDITIVE, AND THE ONE-NUMBER DISPLACED-COST SENTENCE IS THEREFORE UNDERIVABLE RATHER THAN MERELY UNSTATED. Durations are inclusive of callees, so nested producers overlap: on the first run that produced the artifact, summing cross-claim over the shared rows gave 1,850,686ms against the same run's ENTIRE claim-side CPU of 130,335ms -- fourteen times the whole quantity it is supposed to be a part of, and bind_outcome alone reads 424 seconds inclusive, three times the run total by itself. The summary line therefore carries claim_cpu_total_ms as the ceiling any true total must sit under and says cost_columns=inclusive_of_callees_do_not_sum, so the artifact refuses the reading rather than leaving it to a reviewer. NEXT-RUNG TRIGGER, A CAPABILITY: SELF TIME -- inclusive minus the callees the same pass already counted -- after which the column is additive and the sentence is derivable. IT IS A TRACKED STALL RATHER THAN A WISH, because the mechanism already exists one tier over: v1_compiler.v1_interpreter CrossClaimFillGuard's Drop computes exactly that netting against the CROSS_CLAIM_FILL_FRAMES child stack for the shared-fill ledger; what is missing is a child stack over the recompute ledger's frames.\n\nWHAT IT DOES NOT ANSWER, declared rather than left for a reader to discover: it explains the LEVEL and not the VARIANCE. A per-closure constant is by construction identical on two runs of one tree, and a run-to-run delta is measured (32 rows, byte-identical eval_steps, cpu up 1.31-1.80x over a byte-identical payload module -- gentle-wolf-793), so the two compose rather than compete: the constant puts a family AT the line and something run-to-run decides which of its rows cross it, and the second half has two receipts, neither a cross-run comparison: two rows of one module measured 502ms and EXACTLY 500ms against the 500ms budget, and one row was observed crossing the two populations on ONE tree -- run 33620893203 attempt 1 records an_empty_receipt_series_leaves_the_live_tree_unmeasured_rather_than_held as INTERRUPTED-BEFORE-VERDICT with cost=UNMEASURED while ATTEMPT 2 of the same run records it as COMPLETED-OVER-COST-REQUIREMENT at cpu_ms=500 -- a BOUND beside a VALUE rather than two measurements of one quantity, which is why no cost is written for attempt 1: an interrupted row's reported figure names where the poll observed the ceiling and is a property of the budget. At that margin which population a row lands in is decided by whether the poll fired before or after the work finished, which is a fact about the poll rather than about the row. CITE THE ATTEMPT, NOT THE RUN: a bare run id and a bare job id both resolve to the latest attempt, so a rerun silently changes what a citation serves with no error and nothing visible from the citing end -- the readings above are pinned to /actions/runs//attempts//logs, and the other figures in this row come from runs 33615659836 and 33622954427 at run_attempt=1. Through the retraction and the restoration of that evidence, this census reported the same thing throughout, because it keys PRODUCERS and never which rows tipped. The tipping variable is separately open and is not this instrument's subject. Second boundary: the ledger keys PURE NAMED FN evaluations, so a native builtin called directly from a claim body is not a row here -- the same boundary std.evaluation_budget evaluation_budget_opaque_host_call_note draws for the deadline, where no poll stride falls inside an opaque host call and such a claim surfaces as completed-over-cost rather than interrupted. A producer missing from this ranking is therefore not evidence that nothing is re-derived under it.\n\nTHE REDS ARE ENROLLED AND THE FIRST ONE IS THE BLINDNESS ITSELF: a_producer_demanded_once_per_claim_is_invisible_per_frame_and_visible_across_claims asserts duplicated_keys=0 in EACH frame's own ledger beside the census's two-claim row, so re-scoping the census back to one frame reds it. Its controls are a_single_claim_producer_ranks_at_zero_cross_claim_waste (a single-claim cost is the claim's own work and must not score), same_named_producers_at_distinct_declaration_sites_do_not_merge (name-only keying would manufacture cross-claim sharing out of two unrelated single-claim costs) and the_retention_floor_omits_loudly_rather_than_silently.\n\nHAND-ITEM DELTA: the enumerated items are the new module-scope declarations in this diff (v1_interpreter's census types, constants, thread-local and functions, plus the declaration-site helper the unkeyed bucket needed; cli_run's writer). The remaining diff is ExistingSeedItemModified -- EvalRecomputeTrace's unkeyed bucket gains a duration and a declaration site so a composite-argument producer can be RANKED rather than only named, its two call sites follow, and required_floor_runner gains the absorb call, the report block and one env read -- plus enrolled test scope (the four census tests above).", + owning_dissolution_lane: "v1-hand-queue-drain" as RoadmapNodeId, + trigger: "Delete the seed declarations when the claim-frame boundary is a MODELED fact -- the evaluator and its frame lifecycle migrated into the self-emitted substrate -- at which point cross-frame demand is a fold over modeled receipts and this host aggregation has nothing left to stand on. NOT retired by the generic cross-claim pure memo landing: a tier that shares every safely servable pure call removes the RECOMPUTE, and this artifact is the DEMAND measurement that says which values a tier must reach and what its coverage gaps cost -- the two are different facts, and a trigger satisfied by the repair would delete the instrument that measures whether the repair covered anything.", + current_boundary: "v1_compiler.cli_run.required_floor_runner claim loop -> v1_compiler.v1_interpreter absorb_claim_recompute_demand -> cross_claim_demand_rows / cross_claim_demand_disclosure -> [cross-claim-demand] log lines and v1_compiler.cli_run write_required_floor_cross_claim_demand_tsv -> gunbc.witness_floor_workflow required_floor_cross_claim_demand_path artifact" +} diff --git a/dag/gunbc/recurring_failure_mode.dag b/dag/gunbc/recurring_failure_mode.dag index e255f40737d..8162f521ee8 100644 --- a/dag/gunbc/recurring_failure_mode.dag +++ b/dag/gunbc/recurring_failure_mode.dag @@ -237,6 +237,7 @@ data check_subject_narrower_than_its_declared_claim: RecurringFailureMode = Recu authored: "**a check's subject is narrower than the claim it is cited for** (EVERY DEGREE OF FREEDOM IN A CHECK'S SUBJECT -- which call sites, which spellings, which lexical states, which domain, which files -- is a place its coverage can silently be narrower than the population it is read as covering. INVALID STATE: an executing, reached, green check whose declared subject is a strict superset of the population its implementation actually ranges over. RUNG: outside the ladder, because the gap is invisible from the check's own result -- a wall whose range is too narrow is green for the same reason a wall over a clean corpus is green, and DESIGN section 4b(1) reads a class's rung as the MINIMUM across its in-scope paths, so a check like this INFLATES the rung of everything it is cited for. THE ROUTE IS WHAT MAKES IT INVISIBLE: the narrowing is never a decision, it is the shape of the first implementation -- an author edits one file, so the scanner reads one file; the corpus spells the call one way, so the seed matches one spelling; the fixture has no raw strings, so the projection has no raw-string state; the window is never short, so the universal is never vacuous. Each is locally reasonable and none is written down, which is why the gap survives the review that reads the check's LOGIC and never asks what its INPUT was. SPECIMEN, and the reason this is a class rather than a bug: ONE guard in PR #10033 carried SIX instances -- (1) a body-local grep for the mutator, defeated by moving the call one helper deeper; (2) a seed matching only the fully-qualified path, defeated by `use std::env;`; (3) a lexical projection with no raw-string state, so an unbalanced brace in a fixture silently shifted every containment decision after it; (4) `take(n).all(..)`, vacuously true over a short remainder (rostered as predicate_vacuously_true_on_an_empty_domain); (5) a scan ranging over ONE FILE while the scope declaration claimed every test the required lane runs -- found by review, not by the author; and (6) after widening, cross-file call edges resolved by BARE NAME, which merges every declaration of a shared spelling (18 `main`s in that crate) and bridged one production seed into hundreds of unrelated tests. Four were found by the author, one by review, one by the widening itself, and the sixth is the mirror of the fifth: over-broad and under-broad are the same defect measured in opposite directions, and both are reported as coverage. RECOGNITION RULE: name the check's SUBJECT and the check's RANGE as two separate sentences, then ask what is in the first and not the second. If the answer is anything, the claim is the narrower one until the range is widened. The review-side tell is a scope comment that quantifies (`every test`, `all modules`, `any call site`) beside an implementation that names ONE input. BOUNDED AGAINST TWO NEIGHBOURS. selection_view_read_as_population is a set produced by a MEASUREMENT threshold and read as the population, arriving as the fix for a real objection; here nothing is measured and nothing is filtered -- the range was simply never as wide as the sentence beside it. incidental_denominator_as_wall is a guard that holds by an upstream coincidence that genuinely holds; here the guard holds over a population it never read. CEILING 3, structurally guaranteed, and decidable: a check whose subject is a VALUE it receives rather than a scope its implementation re-derives cannot range over less than it claims, because the claim and the range are then one object. NEXT-RUNG TRIGGER, named as a capability: the guarded population is passed to the check as a constructed value carrying its own extent, sufficient that no check can name a subject its input does not contain. WHAT IS NOT CLAIMED: that widening always wins. Narrowing the DECLARATION to match the range is equally honest and is sometimes correct -- what is forbidden is leaving the two different, and the choice between them is a cost measurement, not a preference. On this specimen widening cost one change of input plus one linear-time rewrite of a quadratic edge builder, and the wider check ran FASTER than the narrow one, which is what settled it.)", evidence: [], } +data recurrence_ledger_scoped_below_the_recurrence: RecurringFailureMode = RecurringFailureMode { identity: "recurrence_ledger_scoped_below_the_recurrence" as NonEmptyStr, authored: "**a recurrence ledger scoped below the recurrence it must find** (an instrument counts repetition WITHIN a unit of work and is reset at that unit's boundary, while the repetition that matters happens ACROSS those units. Every unit then reports the same honest answer -- ONE -- and the aggregate nobody computes is the only place the duplication exists. The instrument is not wrong at its own grain and never reads as broken: it runs, it is enabled, it prints, and its output is empty of the thing being looked for, which is indistinguishable from that thing being absent. **The recognition rule is a question about RANGE, not about correctness: name the scope over which the redundancy you are hunting is defined, and check that the ledger's lifetime is at least that scope.** If the ledger is cleared more often than the redundancy recurs, its silence is a property of its reset schedule. **THE KINSHIP THAT MAKES IT LEGIBLE, and it is `censored_estimator_drops_its_own_tail` in a different medium: a cost paid once per unit and attributed to one arbitrary consumer never reaches count>=2 inside that unit, so the ledger that ranks REPETITION cannot observe the population it exists to rank.** The estimator is censored not by a threshold but by a scope. **THE FIELD TELL, and it is the one that identifies this class from the outside before anyone reads the instrument: the repair mechanism ALREADY EXISTS and its coverage is hand-authored, because candidates are discovered when the unrepaired cost causes a failure somewhere unrelated.** A roster that grows one row at a time, each row traceable to an incident on somebody else's change, is a roster whose discovery instrument is blind -- the incidents are doing the ranking, and their cost is paid by whoever was passing. Distinct from `instrument_output_read_as_subject_content`, where a report is complete for the reporter and read past its promise -- here the report IS complete for its declared scope and the CONSUMER'S QUESTION LIVES AT A WIDER ONE -- and distinct from `selection_view_read_as_population`, where a threshold over measured values is mistaken for the class: no selection happens here at all, the population is simply unrepresentable in the ledger's key space. SPECIMEN, WITH THE FULL CAUSAL CHAIN MEASURED (2026-09-02, gunbc#10036 lane). The required floor charges each claim a per-claim CPU ceiling and builds a FRESH EVALUATION FRAME PER CLAIM -- deliberately, so one witness cannot contaminate the next -- so every claim re-derives the pure substrate its import closure reaches. `v1_compiler.v1_interpreter`'s recompute-trace ledger ranks pure calls with count>=2 within ONE `InterpContext`, and `claim_executor` prints and drops it at every claim frame exit; a producer evaluated EXACTLY ONCE per claim, in 3477 claims, has count=1 in every ledger and appears in none. The repair mechanism already existed -- `v2.workflow.floor_pure_producer_share` serves one evaluation of a declared pure producer across frames -- and its roster was six hand-authored rows, because nothing produced its candidates. **SO THE DISCOVERY MECHANISM WAS THE COST: roster coverage was found when a per-claim budget refusal landed on an unrelated lane's pull request, four lanes in one day, every one of those runs reporting failed=0.** THE DISCRIMINATOR THAT PROVED THE CHARGE WAS NOT THE CLAIM'S OWN WORK, and it is reusable: a witness and its DISCRIMINATING RED do materially different work, so when they cost the same to within 2ms while both sit near the ceiling, the cost is not the assertion's -- 296/294, 333/331 and 384/382 on main run 33615659836's `required_floor_claim_cost.tsv`, against a corpus p50 of 2ms; one 34-row module spans 0ms to the ceiling in three bands that track WHICH SHARED PRODUCER a row forces rather than what it asserts. **WHAT THE SPECIMEN REFUTES IS BOTH OF THE OBVIOUS REMEDIES, AND THAT IS WHY THE CLASS IS FILED RATHER THAN THE INSTANCE FIXED.** Raising the ceiling moves a line that the corpus distribution says is correctly placed, to accommodate a charge that is wrong, and it destroys the only signal that anything is slow; quarantining the family converts a cost problem into a coverage problem silently. Both are the absorbing arm DESIGN section 5 forbids, and the second cannot even be reached, because the family is not anomalous -- its own work is among the smallest in the corpus. REPAIR: fold each unit's ledger into a run-scoped census keyed on the DECLARATION identity plus the argument row, so cross-unit demand becomes a run product; the receipt is `v1_compiler.v1_interpreter` `absorb_claim_recompute_demand` and the artifact `gunbc.witness_floor_workflow` `required_floor_cross_claim_demand_path`, whose own discriminating red asserts the per-frame ledger's blindness beside the census's answer so re-scoping it back reds. A SECOND, INDEPENDENT CONSEQUENCE, AND IT RETIRES PART OF A DECLARED DROP RATHER THAN THE DROP: `gunbc.rung_drop` `floor_cost_contention_verdict` names a missing carrier -- the per-claim cost artifact modeled as data a function can read -- and states that its attention subset is a HAND derivation rather than a run product. That is buildable, so what a cross-unit census plus that modeled fold retires is BLIND CANDIDATE DISCOVERY, INCIDENT-AS-DISCOVERY, AND THE MISSING CARRIER. **WHAT IT DOES NOT RETIRE, AND SAYING OTHERWISE WOULD BE THE RUNG INFLATION THIS LEDGER EXISTS TO CATCH: environment-independent claim-cost qualification.** Naming every shared producer does not make a claim's remaining evaluator work invariant across execution envelopes -- host scheduling, contention and cache state still move it -- so the per-claim line stays an ATTEMPT-SAFETY boundary and that drop keeps standing on its own subject. **THE SAME DISCIPLINE APPLIES TO WHAT THE CENSUS ITSELF EXPLAINS.** A per-unit constant is BY CONSTRUCTION identical on two runs of one tree, so it accounts for the LEVEL and not the VARIANCE; a run-to-run delta is separately measured (byte-identical evaluator steps against 1.31-1.80x cpu over a byte-identical payload). The two COMPOSE into the observed shape -- the constant puts a family AT the line and something run-to-run decides which of its rows cross -- and that composition is why the victim WANDERS rather than repeating -- the prediction this account makes and the rival 'fixed set of expensive rows' account does not. **TWO RECEIPTS, NEITHER A CROSS-RUN COMPARISON.** (i) ZERO MARGIN: two rows of one module measured 502ms and EXACTLY 500ms against the 500ms budget. (ii) ONE ROW OBSERVED CROSSING THE TWO POPULATIONS ON ONE TREE -- run 33620893203 attempt 1 records `an_empty_receipt_series_leaves_the_live_tree_unmeasured_rather_than_held` as INTERRUPTED-BEFORE-VERDICT with `cost=UNMEASURED`, and ATTEMPT 2 OF THE SAME RUN records it as COMPLETED-OVER-COST-REQUIREMENT at cpu_ms=500, reaching its verdict. **THAT PAIR IS A BOUND BESIDE A VALUE, NOT TWO MEASUREMENTS OF ONE QUANTITY, AND THE DIFFERENCE IS THE WHOLE POINT:** the interrupted arm knows only that the cost exceeded the ceiling by an unknown amount, and its reported figure names where the POLL observed the ceiling -- a property of the budget, which its own diagnostic spends three clauses saying. Writing the pair as two costs would say the row got a couple of milliseconds cheaper and crossed a line, which is a story about the row; what actually changed between the attempts is WHAT THE OBSERVER COULD SAY. Those are the two populations the budget cut separated BECAUSE THEY ARE DIFFERENT FACTS WITH DIFFERENT REMEDIES, and at this margin which one a row lands in is decided by whether the poll fired before or after the work finished: a fact about the poll, not about the row. **THE CITATION DISCIPLINE THAT MADE THAT RECOVERABLE IS PART OF THE CLASS, because the evidence was retracted and then restored on it: CITE THE ATTEMPT, NOT THE RUN OR THE JOB.** A bare run id and a bare job id both resolve to the LATEST attempt, so a rerun silently changes what a stable citation serves -- no error, and nothing visible from the citing end, which is why the first retraction looked sound. `/actions/runs//attempts//logs` serves the attempt that was actually read; a figure cited without an attempt should carry `run_attempt` beside it so a reader can tell whether the citation still points at what its author saw. **AND THE STANDING RECOMMENDATION IS UNCHANGED BY EITHER DIRECTION OF THAT CORRECTION: A REMEDY MUST NEVER BE KEYED TO WHICH ROWS TIPPED.** A producer census survived the retraction and the restoration without a word changing in what it reports; a ranking of tipped rows would have been rebuilt twice. Two independent arguments for one conclusion: dispersion shows the charge is not the ASSERTION's, wandering shows it is not the ROW's.)", evidence: [] } data recurring_failure_mode_roster: List = [ censored_estimator_drops_its_own_tail, @@ -293,4 +294,5 @@ data recurring_failure_mode_roster: List = [ ambient_process_state_read_by_a_concurrent_reader, predicate_vacuously_true_on_an_empty_domain, check_subject_narrower_than_its_declared_claim, + recurrence_ledger_scoped_below_the_recurrence, ] diff --git a/dag/gunbc/seed_growth_admission.dag b/dag/gunbc/seed_growth_admission.dag index 54a2a0e4343..806a79a1f99 100644 --- a/dag/gunbc/seed_growth_admission.dag +++ b/dag/gunbc/seed_growth_admission.dag @@ -50,6 +50,7 @@ import gunbc.stage0_rust_host_observation { stage0_rust_observation_seed_growth_ import gunbc.floor_non_verdict_enrollment { floor_non_verdict_seed_growth_justification } import gunbc.floor_route_gap_seed_growth { floor_route_gap_seed_growth_justification } import gunbc.cross_claim_pure_share_seed_growth { cross_claim_pure_share_seed_growth_justification } +import gunbc.cross_claim_demand_census_seed_growth { cross_claim_demand_census_seed_growth_justification } import gunbc.frame_independent_symbol_seed_growth { frame_independent_symbol_seed_growth_justification } import gunbc.floor_cost_debt_seed_growth { floor_cost_debt_seed_growth_justification } import gunbc.floor_cost_debt_standing_seed_growth { floor_cost_debt_standing_seed_growth_justification } @@ -193,6 +194,7 @@ fn seed_growth_justification_roster() -> List { floor_route_gap_seed_growth_justification, floor_cost_debt_seed_growth_justification, cross_claim_pure_share_seed_growth_justification, + cross_claim_demand_census_seed_growth_justification, frame_independent_symbol_seed_growth_justification, floor_cost_debt_standing_seed_growth_justification, floor_population_projection_seed_growth_justification, diff --git a/dag/gunbc/witness/witness_floor_workflow.dag b/dag/gunbc/witness/witness_floor_workflow.dag index efd953156d0..143cffbf948 100644 --- a/dag/gunbc/witness/witness_floor_workflow.dag +++ b/dag/gunbc/witness/witness_floor_workflow.dag @@ -927,6 +927,73 @@ data required_floor_claim_cost_path: String = "required_floor_claim_cost.tsv" data required_floor_claim_cost_artifact_name: String = "required-floor-claim-cost" +// THE FIFTH FILE: THE CROSS-CLAIM DEMAND CENSUS, and it answers a question the four above +// structurally cannot. +// +// The per-claim cost receipt says WHAT each claim was charged. It cannot say what the charge was +// FOR, and on this corpus most of a tail row's charge is not its own work: measured on main run +// 33615659836, a witness and its discriminating RED -- which do materially different work -- cost +// within 2ms of each other while both sit near the ceiling, and one 34-row module spans 0ms to the +// ceiling in three bands that track WHICH SHARED PRODUCER a row forces rather than what it +// asserts. That is the floor's fresh-frame-per-claim boundary re-deriving each closure's pure +// substrate once per claim, and DESIGN section 2 prices it as authored duplication whose least +// common ancestor is the RUN. +// +// THE REPAIR ALREADY EXISTS AND ITS DISCOVERY MECHANISM DID NOT. `v2.workflow.floor_pure_producer_share` +// serves one evaluation of a declared pure producer across claim frames -- and its roster is +// hand-authored, because the interpreter's recompute ledger is scoped to ONE evaluation frame and +// reports only keys re-hit inside it. A producer evaluated exactly once per claim, in thousands of +// claims, has count=1 in every frame's ledger and appears in none of them. So the instrument that +// ranks redundant recompute was blind to the entire population the roster is enrolled from, and +// candidates were discovered when a budget refusal landed on an unrelated lane's pull request. +// `absorb_claim_recompute_demand` folds each claim's ledger into a run-scoped census before the +// frame is dropped, and this artifact is that census. +// +// WHAT IT DOES NOT ANSWER, said here because an artifact is read for more than it measures. It +// explains the LEVEL -- why a family sits pinned near the ceiling -- and not the VARIANCE: a +// per-closure constant is the same on two runs of one tree, while 32 rows have been measured with +// byte-identical eval_steps and cpu up 1.31-1.80x across runs over a byte-identical payload. The +// two compose (the constant puts the family AT the line, something run-to-run decides which rows +// cross). Two receipts, neither a cross-run comparison: two rows of one module measured 502ms and +// EXACTLY 500ms against the 500ms budget, and one row was observed crossing the two populations on +// ONE tree -- run 33620893203 attempt 1 records it INTERRUPTED-BEFORE-VERDICT with cost=UNMEASURED +// while attempt 2 of the same run records it COMPLETED-OVER-COST-REQUIREMENT at cpu_ms=500. Which +// population a row lands in at that margin is decided by whether the poll fired before or after +// the work finished. The pair is a BOUND beside a VALUE rather than two measurements of one +// quantity, so no cost is written for attempt 1: an interrupted row's reported figure names where +// the poll observed the ceiling, which is a property of the budget. Both citations pin the ATTEMPT, because a bare run id and a bare job id +// resolve to the latest one and a rerun silently changes what they serve. The tipping +// variable is separately open. And a cost inside a native builtin called directly from a claim +// body is not a row here at all, because the ledger keys pure named fn evaluations -- the same +// boundary std.evaluation_budget's opaque-host-call note draws for the deadline. +// +// THE COST COLUMN IS NOT ADDITIVE AND THE ARTIFACT SAYS SO IN ITS OWN SUMMARY LINE. Durations are +// inclusive of callees, so a producer and everything it calls both appear and their figures +// overlap; summing the column counts the same nanoseconds once per level of nesting. On the first +// run that produced this artifact the naive sum over shared rows was ~1850s against a run whose +// entire claim-side CPU was 130s. So the summary carries claim_cpu_total_ms as the ceiling any +// true total must sit under, and a displaceable-cost figure needs self-time, which the ledger does +// not carry yet. +// +// THE NEGATIVE CONTROL IS EXECUTED RATHER THAN PROMISED, and it is the strongest thing that can +// be said about this artifact's separation of discovery from policy. The two rust target models +// were enrolled in the share roster and REMOVED, because a measured present-versus-absent +// experiment priced their serve above their recompute. They are exactly the kind of row this +// census ranks highly -- and on the first run that produced the artifact, the rust target-model +// chain does rank in the top ten by cross-claim recomputation while remaining unenrolled and +// unenrollable by any path from here. Demand is observed; enrolment stays a separate decision +// with its own measurement, in this order: demand observed, candidate selected, present/absent +// serve experiment on ONE exact tree, portability and argument-reuse evidence, serve cheaper than +// recompute, and only then a roster edit in floor_pure_producer_share. +// +// IT GATES NOTHING AND ENROLS NOTHING. A row is a CANDIDATE whose SERVE cost is unmeasured, and +// `floor_pure_producer_share` records the case where the serve lost to the recompute and the row +// was removed. Arming this variable changes no threshold and no verdict: the writer is a pure +// write of rows the fold already produced, on the same completion path as the four above. +data required_floor_cross_claim_demand_path: String = "required_floor_cross_claim_demand.tsv" + +data required_floor_cross_claim_demand_artifact_name: String = "required-floor-cross-claim-demand" + // THE FLOOR HAS BEEN WRITING THREE FILES ON EVERY RUN AND THROWING ALL THREE AWAY. // // cli_run.rs reads GUNBC_REQUIRED_FLOOR_DISPOSITION, GUNBC_EXPECTED_RED_ROSTER_JOIN and @@ -1129,6 +1196,15 @@ fn witness_floor_bound_steps() -> List { role: capability_neutral, step_name: "Upload the floor's per-claim cost receipt", }, + WitnessFloorBoundStep { + step: witness_floor_tsv_upload_step( + step_name: "Upload the floor's cross-claim demand census", + artifact_name: required_floor_cross_claim_demand_artifact_name, + path: required_floor_cross_claim_demand_path + ), + role: capability_neutral, + step_name: "Upload the floor's cross-claim demand census", + }, WitnessFloorBoundStep { step: witness_toolchain_filesystem_end_probe_step(), role: capability_neutral, @@ -2058,6 +2134,7 @@ data witness_floor_workflow: Workflow = { kv(key: "GUNBC_REQUIRED_FLOOR_DISPOSITION", value: yaml_string(s: required_floor_disposition_path)), kv(key: "GUNBC_LONG_HOME_STORAGE_AGREEMENT", value: yaml_string(s: long_home_storage_agreement_path)), kv(key: "GUNBC_REQUIRED_FLOOR_CLAIM_COST", value: yaml_string(s: required_floor_claim_cost_path)), + kv(key: "GUNBC_REQUIRED_FLOOR_CROSS_CLAIM_DEMAND", value: yaml_string(s: required_floor_cross_claim_demand_path)), kv(key: (required_ci_contract_epoch_env_key as String), value: yaml_string(s: (required_ci_contract_epoch as String))) ] }, diff --git a/docs/design-ledgers.md b/docs/design-ledgers.md index 9d907ec85e7..49ccc87b95d 100644 --- a/docs/design-ledgers.md +++ b/docs/design-ledgers.md @@ -66,6 +66,7 @@ The landing measurement partitions the 31 parser-visible identities into **2 cit - **ambient process state read by a concurrent reader** (a fact that is really a PARAMETER is instead read from a mutable cell the whole process shares -- the working directory, an env var, a global -- while the runtime schedules concurrent readers of it, so the answer a reader gets depends on which writer last ran and the wrong answer is SILENT: the read succeeds, returns a well-formed value, and is simply about a different subject than the reader meant. INVALID STATE: any reader resolving a relative name against ambient state that a concurrent writer may move mid-read. WHY IT SURVIVES REVIEW: each site is locally correct -- it sets the state to the right value before its own work and restores it after -- and the defect exists only in the PRODUCT of sites, which no single diff shows. SPECIMEN (PR #10024, v1_compiler.cli_run): 38 forward set_current_dir sites in one test binary that cargo runs multi-threaded, with the idiom prior = current_dir(); chdir(ws); work; chdir(prior) -- so one test's RESTORE hands the cwd back while a sibling is mid-read. The victim population is not the writers, it is every non-ignored test in the binary that resolves a relative path during the window; the file's own entry_admission_tests note already records the measurement, a DIFFERENT pair of tests failing on each run of identical code (4/1 then 3/2). Seven of the sites were reachable from the required unit lane and in all seven the chdir was DEAD -- every consumer already took an explicit root -- i.e. residue from an unfinished parameterisation, which is the expected shape once the migration is half done. THE CENSUS THAT MATTERS IS REACHABILITY, NOT THE LITERAL AND THE RANGE THAT MATTERS IS THE POPULATION'S, NOT THE FILE'S: the first cut of that gate scanned the one file its author was editing while declaring the whole lane as its subject, which is a selection view read as a population and puts the reported rung above the executed evidence (DESIGN section 4b(1) reads a class's rung as the MINIMUM across in-scope paths). Widening it to every source file of the lane's crate cost a change of input plus a linear-time call extractor -- the per-name matcher is quadratic in lines times declarations and does not survive the wider corpus -- and it surfaced a second trap immediately: resolving cross-file call edges by BARE NAME merges every declaration of a shared spelling, and this crate declares main 18 times, so one production seed bridged into hundreds of unrelated tests. That arm is the absorbing fallback again, in the guard built to catch it. What survives is intra-file edges plus cross-file edges only for names declared exactly once, with the ambiguous remainder pinned by IDENTITY rather than assumed empty. Six forms of the same defect were found in this one guard; they are rostered as their own class at check_subject_narrower_than_its_declared_claim, which this row is a specimen of rather than an authority for: all seven reached the mutation through a helper, so a body-local grep for the call is green over every instance of the class and stays green the moment a writer is moved one call deeper -- the guard must close transitively over call edges. A SECOND, DISTINCT ARM, kept apart because it has a different remedy: memoizing a value DERIVED from the ambient state (workspace_root's OnceLock over a .git-ancestor walk from cwd) freezes whichever writer won the race for the whole process. On this specimen that arm's population is EMPTY -- every forward site targets the checkout root and 37 of 38 force the memo from the pristine cwd one line above their own call -- so it is reported UNSUPPORTED rather than disproven (see a_structural_possibility_is_not_an_occurring_phenomenon); ONE chdir to a non-checkout path arms it, and a fixture that runs git init is such a path. RUNG: found BELOW the ladder, since a redirected read is a silently wrong answer with no typed refusal anywhere. CEILING 4, structurally impossible, and decidable: state passed as a parameter has no shared cell to race on, so the invalid state has no constructor. The intermediate rung is a reachability gate over the guarded population (no_gating_test_reaches_a_process_cwd_mutator), which keeps the state ABSENT but not unwritable; its next-rung trigger is the capability that every reader in the population resolves from an explicitly passed root, after which the ambient write has no consumer and the call can be removed rather than merely counted. REVIEW TELL: a diff that sets shared process state and restores it, in code the runtime may run concurrently -- the restore is the tell, because it is what makes the site look self-contained. - **a predicate vacuously true on an empty domain** (a universally-quantified check -- `all`, `every`, `iter().all(..)`, a for-loop that only ever narrows a flag, a SQL NOT EXISTS -- is satisfied by the IDENTITY ELEMENT when its domain is empty, so the check answers TRUE about a population it never examined. INVALID STATE: a guard whose verdict is read as evidence while its domain is empty, with nothing establishing that the domain was non-empty. WHY IT SURVIVES REVIEW, and this is what separates it from an ordinary off-by-one: the code READS as a check, it EXECUTES, it is REACHED, and on every input a reviewer imagines the domain is non-empty and the predicate genuinely discriminates. The vacuous case is not a value in the data -- IT IS PRODUCED BY A BOUNDARY, so it appears only at the end of a line, the end of a buffer, a remainder shorter than the window, an empty selection, the last chunk. Those are exactly the inputs a hand-written fixture omits, and exactly the inputs a real corpus eventually contains. SPECIMEN (PR #10033, measured): a raw-string scanner tested its closing delimiter with `chars[i+1..].iter().take(hashes).all(|h| *h == '#')`. Over a remainder shorter than `hashes` that iterator yields fewer items and `all` is vacuously TRUE, so a quote near end-of-line closes a hash-delimited raw string that is still open, and the rest of the fixture is handed back to the code scanner -- reintroducing the very brace-depth skew the raw-string handling existed to prevent, silently and in the UNDER-approximating direction. The repair is a length bound, not a different predicate: `i + 1 + hashes <= len` conjoined with the same `all`, which leaves the zero-hash case (a raw string with no hashes) trivially satisfied as it must be. RECOGNITION RULE, and it generalises past scanners: A PREDICATE THAT IS TRUE ON THE EMPTY CASE IS NOT A CHECK UNTIL ITS DOMAIN IS PROVEN NON-EMPTY. Ask of every universal: what does this return when the collection is empty, and can the boundary produce empty? If the answer is TRUE and YES, the domain bound is part of the predicate and its absence is the defect. BOUNDED AGAINST THREE NEIGHBOURS, none of whose recognition rules find it. `executed_conjunct_discriminates_nothing` is about a CORPUS that never constructs the falsifying population, so its remedy is a fixture; here the falsifying population is unconstructable at the boundary by the shape of the quantifier itself, and the remedy is a bound on the domain. `empty_observation_narrow` is an OBSERVATION that could not express what changed being rendered as a verdict -- one layer up, about evidence rather than about a predicate's truth value. `incidental_denominator_as_wall` is a guard that holds by a coincidence upstream that genuinely holds; here nothing holds at all, the guard simply returns TRUE. RUNG: found OUTSIDE the ladder, because the wrong answer is silent and well-formed -- the check reports satisfaction. CEILING 4 and decidable per-site: a windowed comparison whose window length is carried in its type cannot be asked about a short remainder. The intermediate rung is an executed control that exercises the BOUNDARY case specifically, asserted on a synthetic input rather than on the live corpus, so it keeps discriminating power when the corpus changes. - **a check's subject is narrower than the claim it is cited for** (EVERY DEGREE OF FREEDOM IN A CHECK'S SUBJECT -- which call sites, which spellings, which lexical states, which domain, which files -- is a place its coverage can silently be narrower than the population it is read as covering. INVALID STATE: an executing, reached, green check whose declared subject is a strict superset of the population its implementation actually ranges over. RUNG: outside the ladder, because the gap is invisible from the check's own result -- a wall whose range is too narrow is green for the same reason a wall over a clean corpus is green, and DESIGN section 4b(1) reads a class's rung as the MINIMUM across its in-scope paths, so a check like this INFLATES the rung of everything it is cited for. THE ROUTE IS WHAT MAKES IT INVISIBLE: the narrowing is never a decision, it is the shape of the first implementation -- an author edits one file, so the scanner reads one file; the corpus spells the call one way, so the seed matches one spelling; the fixture has no raw strings, so the projection has no raw-string state; the window is never short, so the universal is never vacuous. Each is locally reasonable and none is written down, which is why the gap survives the review that reads the check's LOGIC and never asks what its INPUT was. SPECIMEN, and the reason this is a class rather than a bug: ONE guard in PR #10033 carried SIX instances -- (1) a body-local grep for the mutator, defeated by moving the call one helper deeper; (2) a seed matching only the fully-qualified path, defeated by `use std::env;`; (3) a lexical projection with no raw-string state, so an unbalanced brace in a fixture silently shifted every containment decision after it; (4) `take(n).all(..)`, vacuously true over a short remainder (rostered as predicate_vacuously_true_on_an_empty_domain); (5) a scan ranging over ONE FILE while the scope declaration claimed every test the required lane runs -- found by review, not by the author; and (6) after widening, cross-file call edges resolved by BARE NAME, which merges every declaration of a shared spelling (18 `main`s in that crate) and bridged one production seed into hundreds of unrelated tests. Four were found by the author, one by review, one by the widening itself, and the sixth is the mirror of the fifth: over-broad and under-broad are the same defect measured in opposite directions, and both are reported as coverage. RECOGNITION RULE: name the check's SUBJECT and the check's RANGE as two separate sentences, then ask what is in the first and not the second. If the answer is anything, the claim is the narrower one until the range is widened. The review-side tell is a scope comment that quantifies (`every test`, `all modules`, `any call site`) beside an implementation that names ONE input. BOUNDED AGAINST TWO NEIGHBOURS. selection_view_read_as_population is a set produced by a MEASUREMENT threshold and read as the population, arriving as the fix for a real objection; here nothing is measured and nothing is filtered -- the range was simply never as wide as the sentence beside it. incidental_denominator_as_wall is a guard that holds by an upstream coincidence that genuinely holds; here the guard holds over a population it never read. CEILING 3, structurally guaranteed, and decidable: a check whose subject is a VALUE it receives rather than a scope its implementation re-derives cannot range over less than it claims, because the claim and the range are then one object. NEXT-RUNG TRIGGER, named as a capability: the guarded population is passed to the check as a constructed value carrying its own extent, sufficient that no check can name a subject its input does not contain. WHAT IS NOT CLAIMED: that widening always wins. Narrowing the DECLARATION to match the range is equally honest and is sometimes correct -- what is forbidden is leaving the two different, and the choice between them is a cost measurement, not a preference. On this specimen widening cost one change of input plus one linear-time rewrite of a quadratic edge builder, and the wider check ran FASTER than the narrow one, which is what settled it.) +- **a recurrence ledger scoped below the recurrence it must find** (an instrument counts repetition WITHIN a unit of work and is reset at that unit's boundary, while the repetition that matters happens ACROSS those units. Every unit then reports the same honest answer -- ONE -- and the aggregate nobody computes is the only place the duplication exists. The instrument is not wrong at its own grain and never reads as broken: it runs, it is enabled, it prints, and its output is empty of the thing being looked for, which is indistinguishable from that thing being absent. **The recognition rule is a question about RANGE, not about correctness: name the scope over which the redundancy you are hunting is defined, and check that the ledger's lifetime is at least that scope.** If the ledger is cleared more often than the redundancy recurs, its silence is a property of its reset schedule. **THE KINSHIP THAT MAKES IT LEGIBLE, and it is `censored_estimator_drops_its_own_tail` in a different medium: a cost paid once per unit and attributed to one arbitrary consumer never reaches count>=2 inside that unit, so the ledger that ranks REPETITION cannot observe the population it exists to rank.** The estimator is censored not by a threshold but by a scope. **THE FIELD TELL, and it is the one that identifies this class from the outside before anyone reads the instrument: the repair mechanism ALREADY EXISTS and its coverage is hand-authored, because candidates are discovered when the unrepaired cost causes a failure somewhere unrelated.** A roster that grows one row at a time, each row traceable to an incident on somebody else's change, is a roster whose discovery instrument is blind -- the incidents are doing the ranking, and their cost is paid by whoever was passing. Distinct from `instrument_output_read_as_subject_content`, where a report is complete for the reporter and read past its promise -- here the report IS complete for its declared scope and the CONSUMER'S QUESTION LIVES AT A WIDER ONE -- and distinct from `selection_view_read_as_population`, where a threshold over measured values is mistaken for the class: no selection happens here at all, the population is simply unrepresentable in the ledger's key space. SPECIMEN, WITH THE FULL CAUSAL CHAIN MEASURED (2026-09-02, gunbc#10036 lane). The required floor charges each claim a per-claim CPU ceiling and builds a FRESH EVALUATION FRAME PER CLAIM -- deliberately, so one witness cannot contaminate the next -- so every claim re-derives the pure substrate its import closure reaches. `v1_compiler.v1_interpreter`'s recompute-trace ledger ranks pure calls with count>=2 within ONE `InterpContext`, and `claim_executor` prints and drops it at every claim frame exit; a producer evaluated EXACTLY ONCE per claim, in 3477 claims, has count=1 in every ledger and appears in none. The repair mechanism already existed -- `v2.workflow.floor_pure_producer_share` serves one evaluation of a declared pure producer across frames -- and its roster was six hand-authored rows, because nothing produced its candidates. **SO THE DISCOVERY MECHANISM WAS THE COST: roster coverage was found when a per-claim budget refusal landed on an unrelated lane's pull request, four lanes in one day, every one of those runs reporting failed=0.** THE DISCRIMINATOR THAT PROVED THE CHARGE WAS NOT THE CLAIM'S OWN WORK, and it is reusable: a witness and its DISCRIMINATING RED do materially different work, so when they cost the same to within 2ms while both sit near the ceiling, the cost is not the assertion's -- 296/294, 333/331 and 384/382 on main run 33615659836's `required_floor_claim_cost.tsv`, against a corpus p50 of 2ms; one 34-row module spans 0ms to the ceiling in three bands that track WHICH SHARED PRODUCER a row forces rather than what it asserts. **WHAT THE SPECIMEN REFUTES IS BOTH OF THE OBVIOUS REMEDIES, AND THAT IS WHY THE CLASS IS FILED RATHER THAN THE INSTANCE FIXED.** Raising the ceiling moves a line that the corpus distribution says is correctly placed, to accommodate a charge that is wrong, and it destroys the only signal that anything is slow; quarantining the family converts a cost problem into a coverage problem silently. Both are the absorbing arm DESIGN section 5 forbids, and the second cannot even be reached, because the family is not anomalous -- its own work is among the smallest in the corpus. REPAIR: fold each unit's ledger into a run-scoped census keyed on the DECLARATION identity plus the argument row, so cross-unit demand becomes a run product; the receipt is `v1_compiler.v1_interpreter` `absorb_claim_recompute_demand` and the artifact `gunbc.witness_floor_workflow` `required_floor_cross_claim_demand_path`, whose own discriminating red asserts the per-frame ledger's blindness beside the census's answer so re-scoping it back reds. A SECOND, INDEPENDENT CONSEQUENCE, AND IT RETIRES PART OF A DECLARED DROP RATHER THAN THE DROP: `gunbc.rung_drop` `floor_cost_contention_verdict` names a missing carrier -- the per-claim cost artifact modeled as data a function can read -- and states that its attention subset is a HAND derivation rather than a run product. That is buildable, so what a cross-unit census plus that modeled fold retires is BLIND CANDIDATE DISCOVERY, INCIDENT-AS-DISCOVERY, AND THE MISSING CARRIER. **WHAT IT DOES NOT RETIRE, AND SAYING OTHERWISE WOULD BE THE RUNG INFLATION THIS LEDGER EXISTS TO CATCH: environment-independent claim-cost qualification.** Naming every shared producer does not make a claim's remaining evaluator work invariant across execution envelopes -- host scheduling, contention and cache state still move it -- so the per-claim line stays an ATTEMPT-SAFETY boundary and that drop keeps standing on its own subject. **THE SAME DISCIPLINE APPLIES TO WHAT THE CENSUS ITSELF EXPLAINS.** A per-unit constant is BY CONSTRUCTION identical on two runs of one tree, so it accounts for the LEVEL and not the VARIANCE; a run-to-run delta is separately measured (byte-identical evaluator steps against 1.31-1.80x cpu over a byte-identical payload). The two COMPOSE into the observed shape -- the constant puts a family AT the line and something run-to-run decides which of its rows cross -- and that composition is why the victim WANDERS rather than repeating -- the prediction this account makes and the rival 'fixed set of expensive rows' account does not. **TWO RECEIPTS, NEITHER A CROSS-RUN COMPARISON.** (i) ZERO MARGIN: two rows of one module measured 502ms and EXACTLY 500ms against the 500ms budget. (ii) ONE ROW OBSERVED CROSSING THE TWO POPULATIONS ON ONE TREE -- run 33620893203 attempt 1 records `an_empty_receipt_series_leaves_the_live_tree_unmeasured_rather_than_held` as INTERRUPTED-BEFORE-VERDICT with `cost=UNMEASURED`, and ATTEMPT 2 OF THE SAME RUN records it as COMPLETED-OVER-COST-REQUIREMENT at cpu_ms=500, reaching its verdict. **THAT PAIR IS A BOUND BESIDE A VALUE, NOT TWO MEASUREMENTS OF ONE QUANTITY, AND THE DIFFERENCE IS THE WHOLE POINT:** the interrupted arm knows only that the cost exceeded the ceiling by an unknown amount, and its reported figure names where the POLL observed the ceiling -- a property of the budget, which its own diagnostic spends three clauses saying. Writing the pair as two costs would say the row got a couple of milliseconds cheaper and crossed a line, which is a story about the row; what actually changed between the attempts is WHAT THE OBSERVER COULD SAY. Those are the two populations the budget cut separated BECAUSE THEY ARE DIFFERENT FACTS WITH DIFFERENT REMEDIES, and at this margin which one a row lands in is decided by whether the poll fired before or after the work finished: a fact about the poll, not about the row. **THE CITATION DISCIPLINE THAT MADE THAT RECOVERABLE IS PART OF THE CLASS, because the evidence was retracted and then restored on it: CITE THE ATTEMPT, NOT THE RUN OR THE JOB.** A bare run id and a bare job id both resolve to the LATEST attempt, so a rerun silently changes what a stable citation serves -- no error, and nothing visible from the citing end, which is why the first retraction looked sound. `/actions/runs//attempts//logs` serves the attempt that was actually read; a figure cited without an attempt should carry `run_attempt` beside it so a reader can tell whether the citation still points at what its author saw. **AND THE STANDING RECOMMENDATION IS UNCHANGED BY EITHER DIRECTION OF THAT CORRECTION: A REMEDY MUST NEVER BE KEYED TO WHICH ROWS TIPPED.** A producer census survived the retraction and the restoration without a word changing in what it reports; a ranking of tipped rows would have been rebuilt twice. Two independent arguments for one conclusion: dispersion shows the charge is not the ASSERTION's, wandering shows it is not the ROW's.) ## Declared rung drops (§4b(3)) diff --git a/src/v1/stage0/src/cli_run.rs b/src/v1/stage0/src/cli_run.rs index adb15420a0a..fac5a0f6c60 100644 --- a/src/v1/stage0/src/cli_run.rs +++ b/src/v1/stage0/src/cli_run.rs @@ -39507,6 +39507,60 @@ fn write_required_floor_claim_cost_tsv( Ok(()) } +/// The cross-claim demand receipt: one row per RETAINED producer identity, whether or not more +/// than one claim demanded it. The single-claim rows are kept deliberately — they are the +/// control population that makes `claims > 1` mean something, and dropping them would leave an +/// artifact in which every row looks shared. +/// +/// The header line carries the census's own disclosure: how many claims were folded in, and what +/// the retention floor and the key cap did NOT retain. An artifact that truncates without saying +/// so is read as a population, which is the failure this instrument exists to stop making. +fn write_required_floor_cross_claim_demand_tsv( + path: &str, + rows: &[v1_interpreter::CrossClaimDemandRow], + disclosure: &v1_interpreter::CrossClaimDemandDisclosure, + claim_cpu_total_ms: u128, +) -> Result<(), String> { + use std::io::Write; + let mut file = std::fs::File::create(path) + .map_err(|e| format!("write_required_floor_cross_claim_demand_tsv: create {path}: {e}"))?; + writeln!( + file, + "# summary\tclaims_absorbed={}\tretained_keys={}\tomitted_under_floor={}\tomitted_under_floor_ms={}\tkey_cap_overflow={}\tabsorb_ms={}\tabsorb_max_ms={}\trow_order=identity\tclaim_cpu_total_ms={}\tcost_columns=inclusive_of_callees_do_not_sum", + disclosure.claims_absorbed, + rows.len(), + disclosure.omitted_keys, + disclosure.omitted_ns / 1_000_000, + disclosure.overflow_keys, + disclosure.absorb_ns_total / 1_000_000, + disclosure.absorb_ns_max / 1_000_000, + claim_cpu_total_ms + ) + .map_err(|e| format!("write_required_floor_cross_claim_demand_tsv: write {path}: {e}"))?; + writeln!( + file, + "producer\targ_shape\tclaims\tevals\ttotal_ms\tcross_claim_ms\tmodules\tmodule_sample\tdecl_site" + ) + .map_err(|e| format!("write_required_floor_cross_claim_demand_tsv: write {path}: {e}"))?; + for row in rows { + writeln!( + file, + "{}\t{}\t{}\t{}\t{}\t{}\t{}\t{}\t{}", + row.producer.replace(['\t', '\n'], " "), + row.arg_shape, + row.claims, + row.evals, + row.total_ns / 1_000_000, + row.cross_claim_wasted_ns() / 1_000_000, + row.modules, + row.module_sample.join(",").replace(['\t', '\n'], " "), + row.decl_site.replace(['\t', '\n'], " ") + ) + .map_err(|e| format!("write_required_floor_cross_claim_demand_tsv: write {path}: {e}"))?; + } + Ok(()) +} + fn write_required_floor_disposition_tsv( path: &str, rows: &[RequiredFloorDispositionRow], diff --git a/src/v1/stage0/src/cli_run/required_floor_runner.rs b/src/v1/stage0/src/cli_run/required_floor_runner.rs index 1cb76321b6b..adb5eafd842 100644 --- a/src/v1/stage0/src/cli_run/required_floor_runner.rs +++ b/src/v1/stage0/src/cli_run/required_floor_runner.rs @@ -5039,6 +5039,7 @@ pub fn run_required_floor( let required_floor_disposition_path = std::env::var("GUNBC_REQUIRED_FLOOR_DISPOSITION").ok(); let long_home_storage_agreement_path = std::env::var("GUNBC_LONG_HOME_STORAGE_AGREEMENT").ok(); let claim_cost_path = std::env::var("GUNBC_REQUIRED_FLOOR_CLAIM_COST").ok(); + let cross_claim_demand_path = std::env::var("GUNBC_REQUIRED_FLOOR_CROSS_CLAIM_DEMAND").ok(); let mut roster_join_report = if roster_join_active { // THE DENOMINATOR IS THE ENROLLED ROSTER, NOT THE SURVIVORS. Until 2026-09-01 this took // the post-suppression roster, so the report described only identities the fold could @@ -5378,6 +5379,13 @@ pub fn run_required_floor( let (result, receipt) = run_claim_measured(&frame, &prepared.subject_digest, &claim.qualified); final_symbol_retention = Some(frame.interner_stats_snapshot()); + // FOLD THIS CLAIM'S RECOMPUTE LEDGER INTO THE CROSS-CLAIM CENSUS BEFORE THE FRAME DIES. + // The ledger is per-frame and reports only keys re-hit WITHIN one claim, so a producer + // this closure re-derives exactly once per claim is invisible to it in every claim + // separately — which is the population `v2.workflow.floor_pure_producer_share` is + // enrolled from. Absorbed here, one line after the measurement and before the next + // frame is built, so the census's claim grain is the loop's own. + v1_interpreter::absorb_claim_recompute_demand(&frame, &claim.qualified, &claim.module_path); let claim_rss_after = current_rss_bytes().unwrap_or(0) / 1024; let claim_rss_kb = claim_rss_after.saturating_sub(claim_rss_before); if claim_rss_kb > claim_rss_kb_max { @@ -6639,6 +6647,93 @@ pub fn run_required_floor( if let Some(path) = claim_cost_path { write_required_floor_claim_cost_tsv(&path, &outcome.claim_cost, cost_basis)?; } + // THE CROSS-CLAIM DEMAND CENSUS. The per-claim cost receipt above says WHAT each claim was + // charged; this says WHICH PRODUCER IDENTITY the run re-derived across claims, which is the + // question the charge cannot answer and the one `v2.workflow.floor_pure_producer_share` + // enrolls from. It gates nothing and it enrols nothing: a row here is a candidate whose + // serve cost is still unmeasured, and that roster's own header records the case where the + // serve lost to the recompute. + { + let rows = v1_interpreter::cross_claim_demand_rows(); + let disclosure = v1_interpreter::cross_claim_demand_disclosure(); + // THE CEILING ANY TRUE TOTAL MUST SIT UNDER, carried beside the rows because the cost + // column is inclusive of callees and therefore NOT additive. Without it the first thing a + // reader does is sum the column, which on the first artifact gave ~14x the run's entire + // claim-side CPU. + let claim_cpu_total_ms: u128 = outcome + .claim_cost + .iter() + .map(|row| row.cpu_nanos / 1_000_000) + .sum(); + // A COPY, SORTED FOR A READER. The artifact leaves in identity order; this preview + // orders by cross-claim recomputation and says so in band, so no consumer can read a + // log head as the census's own ranking or as a candidate roster. + let mut shared: Vec<&v1_interpreter::CrossClaimDemandRow> = + rows.iter().filter(|row| row.claims > 1).collect(); + shared.sort_by_key(|row| std::cmp::Reverse(row.cross_claim_wasted_ns())); + eprintln!( + "[cross-claim-demand] claims_absorbed={} retained_keys={} shared_keys={} \ + omitted_under_floor={} omitted_under_floor_ms={} key_cap_overflow={} \ + absorb_ms={} absorb_max_ms={} claim_cpu_total_ms={} (observation only; nothing \ + refuses on these figures. The absorb runs AFTER each claim's measurement returns, so \ + it is outside every claim's charged window and cannot trip the deadline; absorb_ms \ + is what this instrument cost the run in total. DO NOT SUM THE COST COLUMN: durations \ + are inclusive of callees, so a producer and its callees overlap and the sum counts \ + the same nanoseconds once per level of nesting -- claim_cpu_total_ms is the ceiling \ + any true total must sit under.)", + disclosure.claims_absorbed, + rows.len(), + shared.len(), + disclosure.omitted_keys, + disclosure.omitted_ns / 1_000_000, + disclosure.overflow_keys, + disclosure.absorb_ns_total / 1_000_000, + disclosure.absorb_ns_max / 1_000_000, + claim_cpu_total_ms + ); + const CROSS_CLAIM_DEMAND_PRINT_LIMIT: usize = 25; + eprintln!( + "[cross-claim-demand] the {} lines below are a PREVIEW ordered by cross-claim \ + recomputation, not a candidate roster and not the artifact's order: a row is a \ + producer whose SERVE cost is unmeasured, and enrolment is decided only by \ + v2.workflow.floor_pure_producer_share, which has removed a top-ranking pair before \ + on a measured serve-versus-recompute experiment.", + CROSS_CLAIM_DEMAND_PRINT_LIMIT.min(shared.len()) + ); + for row in shared.iter().take(CROSS_CLAIM_DEMAND_PRINT_LIMIT) { + eprintln!( + "[cross-claim-demand] producer={} args={} claims={} evals={} total_ms={} \ + cross_claim_ms={} modules={} sample={} @{}", + row.producer, + row.arg_shape, + row.claims, + row.evals, + row.total_ns / 1_000_000, + row.cross_claim_wasted_ns() / 1_000_000, + row.modules, + row.module_sample.join(","), + row.decl_site + ); + } + if shared.len() > CROSS_CLAIM_DEMAND_PRINT_LIMIT { + eprintln!( + "[cross-claim-demand] ... and {} further shared producer identit(ies), not \ + printed. This list is the {} largest by cross-claim recomputation; the complete \ + retained population is in the cross-claim demand TSV this run uploads, and the \ + omitted-under-floor counters above bound what no artifact retains.", + shared.len() - CROSS_CLAIM_DEMAND_PRINT_LIMIT, + CROSS_CLAIM_DEMAND_PRINT_LIMIT + ); + } + if let Some(path) = cross_claim_demand_path { + write_required_floor_cross_claim_demand_tsv( + &path, + &rows, + &disclosure, + claim_cpu_total_ms, + )?; + } + } if let Some(path) = required_floor_disposition_path { write_required_floor_disposition_tsv( &path, diff --git a/src/v1/stage0/src/v1_interpreter.rs b/src/v1/stage0/src/v1_interpreter.rs index d7d3da69479..c886a3634a7 100644 --- a/src/v1/stage0/src/v1_interpreter.rs +++ b/src/v1/stage0/src/v1_interpreter.rs @@ -2214,6 +2214,205 @@ pub fn warm_cross_claim_pure_producer( }) } +#[cfg(test)] +mod cross_claim_demand_census_tests { + //! THE CENSUS'S OWN EVIDENCE. The discriminating fact is not that a fold aggregates — it is + //! that the aggregate SEES a demand shape the per-frame ledger cannot represent, so each + //! test below asserts the frame ledger's blindness beside the census's answer. Re-scope the + //! census back to one frame and the first test reds; that is what makes it a wall rather + //! than a decoration. + use std::rc::Rc; + + use im::{vector as im_vec, HashMap}; + + use crate::v1_compiler_infer_emit_info::empty_emit_graph_info; + use crate::v1_compiler_infer_items::ResolvedGraph; + use crate::v1_std_core::{make_expr_node, ExprData, SourceSpan}; + + use super::{ + absorb_claim_recompute_demand, clear_cross_claim_demand_census, cross_claim_demand_rows, + eval_recompute_key, eval_recompute_record, trace_totals, ExecutionMode, InterpContext, + Value, + }; + + fn fresh_ctx() -> InterpContext { + let graph = ResolvedGraph { + modules: Rc::new(im_vec![]), + item_registry: Rc::new(HashMap::new()), + diagnostics: Rc::new(im_vec![]), + emit_graph_info: empty_emit_graph_info(), + }; + InterpContext::new(&graph, Rc::new(HashMap::new()), ExecutionMode::Hermetic) + } + + fn node_at(file: &str, start: i64) -> Rc { + make_expr_node( + Rc::new(crate::std_occurrence_identity::NodeOccurrenceIdentity::OccurrenceSynthetic), + Rc::new(ExprData::NoExprData), + Rc::new(im_vec![]), + None, + Rc::new(SourceSpan { + file: file.to_string(), + start, + end: start, + }), + ) + } + + /// Evaluate `producer` once inside one claim's frame, at the given cost, and fold that frame + /// into the census exactly as the floor's claim loop does. + fn one_claim( + producer: &str, + fn_node: &Rc, + arg: i64, + ns: u128, + claim: &str, + module: &str, + ) -> super::EvalRecomputeTotals { + let ctx = fresh_ctx(); + let call_site = node_at("caller.dag", 1); + let args = [(None, Value::Int(arg))]; + let key = eval_recompute_key(&ctx, fn_node, &args).expect("Int args key soundly"); + eval_recompute_record(&ctx, &call_site, fn_node, producer, key, ns); + absorb_claim_recompute_demand(&ctx, claim, module); + let totals = trace_totals(&ctx.eval_recompute_trace.borrow()); + totals + } + + /// THE RED THIS INSTRUMENT EXISTS FOR. One producer, evaluated ONCE in each of two claims — + /// the shape that dominates the required floor's tail cost — is invisible to every frame's + /// own ledger (`duplicated_keys = 0`, because `count = 1` in each), and the census reports + /// it as one identity demanded by two claims across two modules. + #[test] + fn a_producer_demanded_once_per_claim_is_invisible_per_frame_and_visible_across_claims() { + super::refresh_eval_recompute_trace_enabled_cache_for_tests(); + if !super::eval_recompute_trace_enabled() { + std::env::set_var("GUNBC_RECOMPUTE_TRACE", "1"); + super::refresh_eval_recompute_trace_enabled_cache_for_tests(); + } + clear_cross_claim_demand_census(); + let producer_node = node_at("producer.dag", 100); + let first = one_claim( + "tm_target_model", + &producer_node, + 7, + 250_000_000, + "m1.claim_a", + "m1", + ); + let second = one_claim( + "tm_target_model", + &producer_node, + 7, + 250_000_000, + "m2.claim_b", + "m2", + ); + assert_eq!( + (first.duplicated_keys, second.duplicated_keys), + (0, 0), + "the per-frame ledger must report NO duplication — this is the blindness under test, \ + and if it ever reports some, this test is measuring something else" + ); + let rows = cross_claim_demand_rows(); + let row = rows + .iter() + .find(|r| r.producer == "tm_target_model") + .expect("the census must carry the producer both claims demanded"); + assert_eq!(row.claims, 2, "one demand per claim, two claims"); + assert_eq!(row.modules, 2, "two consumer modules"); + assert_eq!( + row.cross_claim_wasted_ns(), + 250_000_000, + "a second claim re-deriving the same identity wastes exactly one derivation" + ); + clear_cross_claim_demand_census(); + } + + /// THE CONTROL THAT KEEPS `claims > 1` MEANINGFUL: a producer demanded by ONE claim is + /// retained and ranks at zero. Its cost is that claim's own work and belongs to the cost + /// receipt; a census that scored it would be reporting every expensive call as shareable. + #[test] + fn a_single_claim_producer_ranks_at_zero_cross_claim_waste() { + super::refresh_eval_recompute_trace_enabled_cache_for_tests(); + if !super::eval_recompute_trace_enabled() { + std::env::set_var("GUNBC_RECOMPUTE_TRACE", "1"); + super::refresh_eval_recompute_trace_enabled_cache_for_tests(); + } + clear_cross_claim_demand_census(); + let only = node_at("producer.dag", 200); + one_claim("tm_only_once", &only, 1, 400_000_000, "m1.claim_a", "m1"); + let rows = cross_claim_demand_rows(); + let row = rows + .iter() + .find(|r| r.producer == "tm_only_once") + .expect("retained, so the shared population has a control to be measured against"); + assert_eq!(row.claims, 1); + assert_eq!(row.cross_claim_wasted_ns(), 0); + clear_cross_claim_demand_census(); + } + + /// TWO DECLARATIONS SPELLED THE SAME ARE TWO PRODUCERS. Keying on the name alone would merge + /// them into one row reporting a producer that does not exist — and it would do so in the + /// direction that manufactures cross-claim sharing out of two unrelated single-claim costs. + #[test] + fn same_named_producers_at_distinct_declaration_sites_do_not_merge() { + super::refresh_eval_recompute_trace_enabled_cache_for_tests(); + if !super::eval_recompute_trace_enabled() { + std::env::set_var("GUNBC_RECOMPUTE_TRACE", "1"); + super::refresh_eval_recompute_trace_enabled_cache_for_tests(); + } + clear_cross_claim_demand_census(); + let here = node_at("alpha.dag", 10); + let there = node_at("beta.dag", 10); + one_claim("tm_homonym", &here, 1, 200_000_000, "m1.claim_a", "m1"); + one_claim("tm_homonym", &there, 1, 200_000_000, "m2.claim_b", "m2"); + let rows: Vec<_> = cross_claim_demand_rows() + .into_iter() + .filter(|r| r.producer == "tm_homonym") + .collect(); + assert_eq!(rows.len(), 2, "two declarations, two rows"); + assert!( + rows.iter().all(|r| r.claims == 1), + "neither is shared: merging them would invent cross-claim demand" + ); + clear_cross_claim_demand_census(); + } + + /// THE RETENTION FLOOR DISCLOSES WHAT IT DROPS. A sub-millisecond first sighting is not + /// retained at identity grain, and the omitted count and its summed cost are reported — an + /// artifact that truncated silently would be read as the population. + #[test] + fn the_retention_floor_omits_loudly_rather_than_silently() { + super::refresh_eval_recompute_trace_enabled_cache_for_tests(); + if !super::eval_recompute_trace_enabled() { + std::env::set_var("GUNBC_RECOMPUTE_TRACE", "1"); + super::refresh_eval_recompute_trace_enabled_cache_for_tests(); + } + clear_cross_claim_demand_census(); + let tiny = node_at("producer.dag", 300); + one_claim("tm_tiny", &tiny, 1, 1_000, "m1.claim_a", "m1"); + let d = super::cross_claim_demand_disclosure(); + assert_eq!(d.claims_absorbed, 1); + assert_eq!(d.overflow_keys, 0); + assert_eq!( + d.omitted_keys, 1, + "the sub-floor key is counted, not forgotten" + ); + assert_eq!( + d.omitted_ns, 1_000, + "and its cost is carried in the disclosure" + ); + assert!( + !cross_claim_demand_rows() + .iter() + .any(|r| r.producer == "tm_tiny"), + "it is genuinely not retained — the disclosure is the only place it exists" + ); + clear_cross_claim_demand_census(); + } +} + #[cfg(test)] mod cross_claim_memo_tests { use crate::v1_rt::RcStr; @@ -3164,7 +3363,11 @@ struct ParseTableMemo { #[derive(Default)] struct EvalRecomputeTrace { map: std::collections::HashMap, - unkeyed_by_fn: std::collections::HashMap, + // (calls, inclusive nanos, declaration site) per composite-argument producer. The nanos + // half is here for the CROSS-CLAIM census below: a producer whose arguments have no cheap + // sound identity is exactly as capable of costing a claim 300ms once as a nullary one, and + // a bucket that counted calls without duration could name it and never rank it. + unkeyed_by_fn: std::collections::HashMap, // fn-node Rcs kept alive so fn_ptr keys stay valid for the ctx lifetime // (same discipline as PureCallMemo.keepalive_fns). keepalive_fns: Vec>, @@ -7124,10 +7327,16 @@ fn eval_pure_named_call( let key = match eval_recompute_key(ctx, fn_node, args) { Some(key) => key, None => { - if trace_on { - eval_recompute_record_unkeyed(ctx, func_name); + if !trace_on { + return call_function(ctx, fn_node, args, env); } - return call_function(ctx, fn_node, args, env); + // TIMED, not merely counted: the composite-argument bucket feeds the cross-claim + // census, which ranks by duration. Recording after the call is what makes the two + // buckets comparable; the earlier count-only form could name a producer it could + // never rank. + let result = call_function(ctx, fn_node, args, env); + eval_recompute_record_unkeyed(ctx, fn_node, func_name, started.elapsed().as_nanos()); + return result; } }; if memo_on { @@ -7793,10 +8002,29 @@ fn eval_recompute_record( } } -fn eval_recompute_record_unkeyed(ctx: &InterpContext, func_name: &str) { +fn eval_recompute_record_unkeyed( + ctx: &InterpContext, + fn_node: &Rc, + func_name: &str, + elapsed_ns: u128, +) { let mut t = ctx.eval_recompute_trace.borrow_mut(); t.unkeyed_calls += 1; - *t.unkeyed_by_fn.entry(func_name.to_string()).or_insert(0) += 1; + let row = t + .unkeyed_by_fn + .entry(func_name.to_string()) + .or_insert_with(|| (0, 0, eval_recompute_decl_site(fn_node))); + row.0 += 1; + row.1 += elapsed_ns; +} + +/// A producer's DECLARATION site, `file:offset`. The cross-claim census below keys on it beside +/// the name because `func_name` is the call's spelling: two modules may declare the same bare +/// name, and merging them would report one producer that does not exist. Unlike `fn_ptr`, a span +/// is derived from the source and is therefore equal across the fresh evaluation frame the floor +/// builds per claim — which is the whole boundary this census has to cross. +fn eval_recompute_decl_site(fn_node: &Rc) -> String { + format!("{}:{}", fn_node.span.file, fn_node.span.start) } /// Print the recompute-trace ledger to stderr. No-op unless GUNBC_RECOMPUTE_TRACE=1. @@ -7853,12 +8081,15 @@ pub fn print_eval_recompute_trace(ctx: &InterpContext) { site_labels.join(" @") ); } - let mut unkeyed: Vec<(&String, &u64)> = t.unkeyed_by_fn.iter().collect(); - unkeyed.sort_by(|a, b| b.1.cmp(a.1)); - for (name, count) in unkeyed.iter().take(10) { + let mut unkeyed: Vec<(&String, &(u64, u128, String))> = t.unkeyed_by_fn.iter().collect(); + unkeyed.sort_by(|a, b| b.1 .1.cmp(&a.1 .1)); + for (name, (count, ns, site)) in unkeyed.iter().take(10) { eprintln!( - "[recompute-trace] unkeyed fn={} calls={} (composite args — identity not tracked in slice 1)", - name, count + "[recompute-trace] unkeyed fn={} calls={} total_ms={} @{} (composite args — identity not tracked in slice 1)", + name, + count, + ns / 1_000_000, + site ); } let (hits, misses, overflow) = eval_call_memo_counters(ctx); @@ -7873,6 +8104,449 @@ fn entry_wasted_ns(e: &EvalRecomputeEntry) -> u128 { e.total_ns - e.total_ns / u128::from(e.count) } +// ── THE CROSS-CLAIM DEMAND CENSUS ──────────────────────────────────────────────────────────── +// +// WHAT IT IS FOR, stated first because the mechanism is small and the reason is the whole point: +// it names, as a RUN PRODUCT, the pure producers the required floor re-derives once per claim +// across many claims — the candidate population for `v2.workflow.floor_pure_producer_share`. +// +// THE BLINDNESS IT CLOSES. `EvalRecomputeTrace` is scoped to one `InterpContext` and reports +// keys with `count >= 2`; the floor builds a FRESH FRAME PER CLAIM and prints and drops the +// ledger at every claim boundary. So a producer evaluated EXACTLY ONCE per claim, in thousands +// of claims, has `count = 1` in every ledger and appears in none of them — the instrument that +// exists to rank redundant recompute cannot see the one recurrence shape that dominates the +// floor's per-claim cost. Measured on main run 33615659836 (`required_floor_claim_cost.tsv`, +// 3477 rows): a witness and its discriminating RED, which do materially different work, cost +// within 2ms of each other at 294/296, 331/333 and 382/384 while p50 for the corpus is 2ms. A +// cost that does not move when the assertion changes is not the assertion's cost; it is the +// closure's fixed re-derivation, and DESIGN §2 prices it as authored duplication whose least +// common ancestor is the RUN. This census is the demand side of that: it does not share, cache +// or enrol anything — it reports which identity is being re-demanded across the frame boundary, +// so roster candidacy stops being discovered by a red landing on an unrelated lane's PR. +// +// IT IS AN OBSERVATION AND NEVER A GATE. Nothing refuses on these figures. Durations are +// wall-of-the-calling-thread inclusive of callees and are as environment-sensitive as every +// other cost reading on a shared runner (`gunbc.rung_drop floor_cost_contention_verdict`); the +// CLAIM COUNT is the stable half, and it is a property of the corpus and the plan rather than +// of the machine. +// +// THE COST COLUMN IS NOT ADDITIVE, AND THIS PARAGRAPH EXISTS BECAUSE THE ARTIFACT INVITES THE +// ERROR. Durations are INCLUSIVE OF CALLEES — the ledger times a producer's whole subtree — so a +// producer and everything it calls both appear, and their figures OVERLAP. Summing the column +// therefore counts the same nanoseconds once per level of nesting. Measured on the first run that +// produced the artifact: summing cross-claim over the shared rows gives ~1850s against a run +// whose ENTIRE claim-side CPU was 130s — a 14x overcount, which is the nesting and nothing else. +// So a per-row figure is a valid statement about THAT producer, and no sum of rows is a valid +// statement about the run. The run's own total claim CPU travels in the summary line as the +// ceiling any true total must sit under. +// +// SO THE DISPLACED-COST SENTENCE IS UNDERIVABLE TODAY, NOT MERELY UNSTATED, and the difference is +// the whole reason this paragraph is a trigger rather than a caveat. "This floor recomputes N +// seconds of pure producer work per run" is the sentence that makes the stake legible, and NO +// arithmetic over this artifact produces it: the quantity it needs does not exist in the column. +// +// NEXT-RUNG TRIGGER, AS A CAPABILITY: SELF TIME — a producer's inclusive duration minus the +// callees the same pass already counted. With it the column is ADDITIVE, the sum becomes a true +// statement about the run, and the displaced-cost sentence is derivable. WHERE IT ALREADY LIVES, +// so this is a tracked stall and not a wish: `CrossClaimFillGuard`'s `Drop` in this file computes +// exactly that quantity one tier over — `inclusive_cpu.saturating_sub(children_cpu)` against the +// `CROSS_CLAIM_FILL_FRAMES` child stack — for the shared-fill ledger. The mechanism exists; what +// is missing is a child stack over the RECOMPUTE ledger's frames. Deliberately not built here: +// this bridge is what stops the next four lanes paying, and a new measurement tier would put it +// behind a fresh review cycle on the fleet condition it exists to explain. +// +// AN OUT-OF-SCOPE OBSERVATION THE CEILING COLUMN MADE, RECORDED HERE BECAUSE ITS ONLY OTHER HOME +// IS TWO LOGS THAT AGE OUT. `claim_cpu_total_ms` — added purely as a bound against summing the +// cost column — measured 130335 on run 33615659836 and 99943 on run 33631458679 (both +// `run_attempt=1`), over the same corpus. A THIRTY PERCENT SWING IN A RUN'S TOTAL CLAIM CPU, and +// a per-closure constant cannot produce it: the constant is by construction identical on two runs +// of one tree. It is the same variance term measured independently as byte-identical evaluator +// steps against 1.31-1.80x cpu, reached here from the opposite direction and at whole-run grain. +// UNEXPLAINED AND DELIBERATELY NOT PURSUED HERE — this census's subject is the LEVEL — but it is +// a property of the ENVIRONMENT measured from inside the run, which is what a fleet-variance +// account has so far lacked. THE TWO FIGURES ARE TRANSCRIBED, as a declared exception to naming +// the instrument instead of copying its output, for the reason the floor-cost carrier grants the +// same exception: THE SUBJECT IS THE DIVERGENCE BETWEEN TWO RUNS, which no single producer on +// this side of the boundary can re-derive once the logs expire. +// +// WHAT THIS CENSUS CANNOT SEE, DECLARED HERE RATHER THAN DISCOVERED BY A LATER READER. Two +// boundaries, and neither is a defect in the aggregation — they are properties of the ledger it +// folds, and a reader who does not know them will read absence as evidence. +// +// (1) IT EXPLAINS THE LEVEL, NOT THE VARIANCE. A per-closure constant is BY CONSTRUCTION the +// same on two runs of the same tree, so it cannot produce a run-to-run delta — and one is +// measured: 32 rows with byte-identical `eval_steps` and cpu up 1.31-1.80x between two runs over +// a byte-identical payload module (gentle-wolf-793, 2026-09-02). The two compose into the shape +// the floor keeps showing: the constant puts a family AT the line, and something run-to-run +// decides WHICH of its rows cross it, and there are TWO INDEPENDENT RECEIPTS for that, neither of +// which is a cross-run comparison. (i) ZERO MARGIN, single attempt: two rows of one module +// COMPLETED over budget — so these are measured values, not preemption bounds — at 502ms and +// EXACTLY 500ms against the 500ms budget. (ii) A ROW OBSERVED CROSSING THE +// TWO POPULATIONS ON ONE TREE: run 33620893203, attempt 1, records +// `test.claim.self_host_compile_phase_live_gate_witness.an_empty_receipt_series_leaves_the_live_tree_unmeasured_rather_than_held` +// as INTERRUPTED-BEFORE-VERDICT with `cost=UNMEASURED`, and ATTEMPT 2 of the SAME RUN records it +// as COMPLETED-OVER-COST-REQUIREMENT at cpu_ms=500 — reaching its verdict. Those are the two +// populations the 2026-08-19 budget cut separated BECAUSE THEY ARE DIFFERENT FACTS WITH DIFFERENT +// REMEDIES, and at this margin which one a row lands in is decided by whether the poll fired +// before or after the work finished. A fact about the poll, not about the row. +// +// AND THE PAIR IS NOT TWO MEASUREMENTS OF ONE QUANTITY — one is a BOUND and one is a VALUE, which +// is why no cost figure is written for attempt 1 here. The interrupted arm knows only that the +// cost exceeded 500ms by an unknown amount; its `interrupt_point` names where the poll observed +// the ceiling, so it is a property of the BUDGET and the diagnostic says so in three clauses. +// Writing it as "502ms in attempt 1 against 500ms in attempt 2" would say the row got two +// milliseconds cheaper and crossed a line — a story about the row, and the instrument-property- +// as-subject-property misreading this very census exists to stop. What changed between the +// attempts is what the OBSERVER could say. +// +// CITE THE ATTEMPT, NOT THE RUN. Both readings above are pinned to +// `/actions/runs//attempts//logs`, because a bare run id and a bare job id BOTH resolve to +// the latest attempt: a rerun silently changes what a stable citation serves, with no error and +// nothing visible from the citing end. A near-disjointness reading was retracted and then +// partially restored on exactly that discovery. This file's other figures come from run +// 33615659836 and run 33622954427, both `run_attempt=1` at the time of writing — recorded so a +// reader can tell whether the citation still points at what was read. Neither half alone predicts +// the wandering — a constant alone gives +// the same rows every run, and runner noise alone over a corpus with p50=2ms tips nobody. This +// census addresses the charge; the tipping variable is separately open and is not this +// instrument's subject. +// +// (2) A COST THAT LIVES INSIDE A NATIVE BUILTIN CALLED DIRECTLY IS NOT A ROW HERE. The ledger +// keys PURE NAMED FN evaluations; a `free_call.*` arm dispatched straight to Rust from a claim +// body is not one, so its cost is invisible to this census — while the same work reached THROUGH +// a .dag producer is captured, because the producer's own timing spans it. `std.evaluation_budget` +// `evaluation_budget_opaque_host_call_note` records the matching fact for the deadline: no poll +// stride falls inside an opaque host call, so such a claim cannot be interrupted at all and +// surfaces as completed-over-cost instead. So a producer missing from this ranking is not +// evidence that nothing is re-derived under it. +// +// ADMISSION TO THIS CENSUS IS NOT ADMISSION TO THE SHARE ROSTER. A row here is a CANDIDATE. +// `v2.workflow.floor_pure_producer_share` records the criterion that decides enrolment and the +// measurement that refuted the obvious case: the two rust target models were enrolled and +// REMOVED because SERVING them costs more than recomputing them. So this artifact ranks demand; +// it does not price the serve, and reading a top row as an enrolment instruction would trade a +// measured red for an unmeasured regression. + +/// One producer identity, aggregated across claim frames. +#[derive(Clone)] +pub struct CrossClaimDemandRow { + /// The call's spelling. + pub producer: String, + /// `file:offset` of the DECLARATION. Frame-independent, and the disambiguator that keeps two + /// same-named producers in different modules from merging into one row that does not exist. + pub decl_site: String, + /// `keyed` — one row per (declaration, argument row) with sound argument identity — or + /// `unkeyed`, the composite-argument bucket, where one row covers ALL argument rows of that + /// declaration and the claim count is therefore an upper bound on any single identity's. + pub arg_shape: &'static str, + /// Distinct claims in which this identity was evaluated at least once. + pub claims: u64, + /// Evaluations across all of them. + pub evals: u64, + /// Inclusive nanos across all of them. + pub total_ns: u128, + /// Distinct consumer modules, and a bounded sample of their names. + pub modules: u64, + pub module_sample: Vec, +} + +impl CrossClaimDemandRow { + /// What a provider whose scope reached the RUN would have saved: everything beyond one + /// evaluation's amortized share across the claims that demanded it. Zero for a producer + /// demanded by exactly one claim, which is the point — a single-claim cost is a claim's own + /// work and belongs to the cost receipt, not to this census. + pub fn cross_claim_wasted_ns(&self) -> u128 { + if self.claims <= 1 { + return 0; + } + self.total_ns - self.total_ns / u128::from(self.claims) + } +} + +#[derive(PartialEq, Eq, Hash, Clone)] +struct CrossClaimDemandKey { + producer: String, + decl_site: String, + /// THE CANONICAL ARGUMENT ROW, NOT A DIGEST OF IT. A 64-bit hash stood here and it could + /// merge two distinct argument rows into one row reporting cross-claim demand that never + /// happened — a fabricated identity, which DESIGN §5 forbids outright and which the header + /// above simultaneously promised not to do ("sound argument identity"). The ledger already + /// keys on this exact vector; carrying it costs the memory the retention floor and key cap + /// are here to bound, and it makes the collision unrepresentable rather than unlikely. + /// Found by review 58673. + args: Vec, + /// `keyed` and `unkeyed` are DIFFERENT SUBJECTS and must not share a key: a nullary keyed + /// call and the composite-argument bucket of the same declaration both carry an empty + /// argument vector, and merging them would sum one identity's cost with all of another's. + arg_shape: &'static str, +} + +struct CrossClaimDemandCell { + claims: u64, + evals: u64, + total_ns: u128, + last_claim: String, + /// THE COMPLETE consumer-module set, interned so one name costs one allocation for the whole + /// census. It is complete because the reported count must be exact: the earlier form kept a + /// bounded SAMPLE and counted against it, so once the sample filled, every later claim from + /// an unsampled module incremented the count again and the column reported CLAIM OCCURRENCES + /// while promising distinct modules. Bounding and counting cannot share one container. + /// Found by review 58673. + modules: std::collections::BTreeSet>, +} + +/// Rows below this are not retained at identity grain. The census's subject is a producer whose +/// per-claim cost is a visible share of a 500ms ceiling; a sub-millisecond frame cost cannot be +/// one however often it recurs at this grain, and retaining every such key across thousands of +/// claims is the memory the floor does not have. THE OMITTED MASS IS DISCLOSED — count and +/// summed nanos — because an artifact that truncates silently is read as a population +/// (`gunbc.recurring_failure_mode instrument_output_read_as_subject_content`). +const CROSS_CLAIM_DEMAND_RETENTION_FLOOR_NS: u128 = 1_000_000; + +/// Distinct census keys retained. Overflow is counted and disclosed, never silent. +const CROSS_CLAIM_DEMAND_KEY_CAP: usize = 200_000; + +/// Consumer-module names RENDERED per row. The count is exact and comes from the complete set; +/// this bounds only how many names a row shows. +const CROSS_CLAIM_DEMAND_MODULE_SAMPLE_CAP: usize = 8; + +#[derive(Default)] +struct CrossClaimDemandCensus { + map: std::collections::HashMap, + claims_absorbed: u64, + omitted_keys: u64, + omitted_ns: u128, + overflow_keys: u64, + /// One `Rc` per distinct module name, shared by every row that names it. + module_names: std::collections::HashMap>, + /// WHAT THE INSTRUMENT ITSELF COSTS, measured rather than argued. A discovery instrument that + /// materially raised every claim's cost would recreate the incident it explains, on a floor + /// that preempts on marginal cost. Two facts make that checkable rather than assumed: the + /// absorb runs AFTER `run_claim_measured` returns, so it is outside the measured window and + /// cannot enter any claim's charged clocks or trip the deadline — and its own total and worst + /// single claim are reported beside the census, so "outside the window" is not asked to stand + /// alone. + absorb_ns_total: u128, + absorb_ns_max: u128, +} + +thread_local! { + /// Process-lifetime on the thread that runs the claim loop. The floor executes its claims + /// sequentially in one process over one prepared subject, which is exactly the scope this + /// census is about; a run whose claims were distributed would observe only its own share and + /// the artifact would say so by its claim count rather than by a silent partial answer. + static CROSS_CLAIM_DEMAND: RefCell = + RefCell::new(CrossClaimDemandCensus::default()); +} + +fn cross_claim_demand_absorb_one( + census: &mut CrossClaimDemandCensus, + key: CrossClaimDemandKey, + evals: u64, + total_ns: u128, + claim: &str, + module_path: &str, +) { + if total_ns < CROSS_CLAIM_DEMAND_RETENTION_FLOOR_NS && !census.map.contains_key(&key) { + census.omitted_keys += 1; + census.omitted_ns += total_ns; + return; + } + if !census.map.contains_key(&key) && census.map.len() >= CROSS_CLAIM_DEMAND_KEY_CAP { + census.overflow_keys += 1; + return; + } + let module = match census.module_names.get(module_path) { + Some(name) => name.clone(), + None => { + let name: Rc = Rc::from(module_path); + census + .module_names + .insert(module_path.to_string(), name.clone()); + name + } + }; + let cell = census + .map + .entry(key) + .or_insert_with(|| CrossClaimDemandCell { + claims: 0, + evals: 0, + total_ns: 0, + last_claim: String::new(), + modules: std::collections::BTreeSet::new(), + }); + // One claim contributes ONE to `claims` however many times it evaluated the identity — + // within-frame repetition is the existing ledger's subject and double-counting it here would + // make an intra-claim loop look like cross-claim recurrence. + if cell.last_claim != claim { + cell.claims += 1; + cell.last_claim = claim.to_string(); + // A set insert, so a module already present is not counted twice however many of its + // claims arrive — the count is `len()` of the complete set and never an incremented + // tally that can drift from it. + cell.modules.insert(module); + } + cell.evals += evals; + cell.total_ns += total_ns; +} + +/// Fold ONE finished claim's recompute ledger into the cross-claim census, before the claim's +/// frame is dropped. Call it after the claim has run and before the next frame is built; a +/// no-op unless `GUNBC_RECOMPUTE_TRACE=1`, which the floor sets for itself. +/// +/// THE CALLER MUST OWN A FRESH FRAME PER CLAIM, and that is a real precondition rather than a +/// style note. The ledger accumulates for the lifetime of its `InterpContext` and is NOT +/// cleared by `eval_call_memo_frame_exit`, so a surface that shares one context across several +/// claims — `claim_batch` shares one per ENTRY — would fold each claim's ledger again on the +/// next call, inflating both `evals` and `total_ns` and, worse, reporting cross-claim demand +/// where there is only one frame's. The required floor builds a frame per claim +/// (`required_floor_runner`, "FRESH PER CLAIM"), which is exactly the boundary this census +/// exists to measure across, so it is the only caller today. A shared-context surface wanting +/// this census needs a per-claim ledger reset first; getting that wrong would commit the class +/// this instrument was built to close, one grain up. +pub fn absorb_claim_recompute_demand(ctx: &InterpContext, claim: &str, module_path: &str) { + if !eval_recompute_trace_enabled() { + return; + } + let started = Instant::now(); + let t = ctx.eval_recompute_trace.borrow(); + // One declaration-site lookup per fn pointer for this frame, rather than a linear scan of + // the keepalive vector per ledger key: the map holds one entry per (fn, argument row) and + // the vector one per fn, so the scan was quadratic in a frame's producer count -- the + // cost-shape defect DESIGN section 6 says is always fixed, inside an instrument whose whole + // subject is cost. + let mut decl_sites: std::collections::HashMap = std::collections::HashMap::new(); + for node in t.keepalive_fns.iter() { + decl_sites + .entry(Rc::as_ptr(node) as usize) + .or_insert_with(|| eval_recompute_decl_site(node)); + } + CROSS_CLAIM_DEMAND.with(|c| { + let mut census = c.borrow_mut(); + census.claims_absorbed += 1; + for (key, entry) in t.map.iter() { + cross_claim_demand_absorb_one( + &mut census, + CrossClaimDemandKey { + producer: entry.fn_name.to_string(), + decl_site: decl_sites.get(&key.fn_ptr).cloned().unwrap_or_default(), + args: key.args.clone(), + arg_shape: "keyed", + }, + entry.count, + entry.total_ns, + claim, + module_path, + ); + } + for (name, (calls, ns, site)) in t.unkeyed_by_fn.iter() { + cross_claim_demand_absorb_one( + &mut census, + CrossClaimDemandKey { + producer: name.clone(), + decl_site: site.clone(), + args: Vec::new(), + arg_shape: "unkeyed", + }, + *calls, + *ns, + claim, + module_path, + ); + } + let spent = started.elapsed().as_nanos(); + census.absorb_ns_total += spent; + if spent > census.absorb_ns_max { + census.absorb_ns_max = spent; + } + }); +} + +/// The census in deterministic identity order. EVERY retained row is returned, +/// single-claim rows included — they rank at zero and they are the control population that makes +/// `claims > 1` mean something, so filtering them here would leave an artifact in which every +/// row looks shared. Selecting the shared ones is the READER's move and it is a view rather than +/// the population: the runner applies it at the print site and the TSV keeps both. +/// +/// (An earlier revision of this sentence said single-claim rows were dropped here. They never +/// were, the tests and the writer both assert they are not, and the sentence pointed a reader at +/// the wrong side of the census's one load-bearing claim. Found by review 58659.) +pub fn cross_claim_demand_rows() -> Vec { + CROSS_CLAIM_DEMAND.with(|c| { + let census = c.borrow(); + let mut rows: Vec = census + .map + .iter() + .map(|(key, cell)| CrossClaimDemandRow { + producer: key.producer.clone(), + decl_site: key.decl_site.clone(), + arg_shape: key.arg_shape, + claims: cell.claims, + evals: cell.evals, + total_ns: cell.total_ns, + // EXACT, read from the complete set. The bound below is on how many NAMES a + // reader is shown and it cannot reach the count. + modules: cell.modules.len() as u64, + module_sample: cell + .modules + .iter() + .take(CROSS_CLAIM_DEMAND_MODULE_SAMPLE_CAP) + .map(|m| m.to_string()) + .collect(), + }) + .collect(); + // IDENTITY ORDER, NOT COST ORDER, and that is a DESIGN section 3 distinction rather than + // a taste. An ordering by cost is a RANKING, and a ranking is a judgment about which + // demands matter -- meaning, which belongs to .dag folding this artifact, not to the seed + // whose warrant here is carrying facts across a frame boundary only the seed can reach. + // So rows leave in a deterministic identity order with every cost column beside them, and + // the runner's log preview sorts a COPY and says in band that it is a preview. + rows.sort_by(|a, b| { + a.producer + .cmp(&b.producer) + .then_with(|| a.decl_site.cmp(&b.decl_site)) + .then_with(|| a.arg_shape.cmp(b.arg_shape)) + .then_with(|| b.total_ns.cmp(&a.total_ns)) + }); + rows + }) +} + +/// What the census did not retain, and what it cost to run. Both halves travel with the rows, +/// because either one read alone invites a wrong conclusion: the omissions bound what the ranking +/// cannot contain, and the overhead answers whether a discovery instrument on a cost-preempting +/// floor is paying for itself. +pub struct CrossClaimDemandDisclosure { + pub claims_absorbed: u64, + pub omitted_keys: u64, + pub omitted_ns: u128, + pub overflow_keys: u64, + pub absorb_ns_total: u128, + pub absorb_ns_max: u128, +} + +/// The census's own statement of what it did not retain and what it cost. Every consumer of the +/// rows above must report this beside them. +pub fn cross_claim_demand_disclosure() -> CrossClaimDemandDisclosure { + CROSS_CLAIM_DEMAND.with(|c| { + let census = c.borrow(); + CrossClaimDemandDisclosure { + claims_absorbed: census.claims_absorbed, + omitted_keys: census.omitted_keys, + omitted_ns: census.omitted_ns, + overflow_keys: census.overflow_keys, + absorb_ns_total: census.absorb_ns_total, + absorb_ns_max: census.absorb_ns_max, + } + }) +} + +/// Reset — for tests, and for any caller that runs two independent floors in one process. +pub fn clear_cross_claim_demand_census() { + CROSS_CLAIM_DEMAND.with(|c| *c.borrow_mut() = CrossClaimDemandCensus::default()); +} + /// Ledger totals for one InterpContext — the materialization demand receipt at /// the eval-frame grain. Key counts are deterministic for a fixed corpus and /// entry set; wasted_ns durations are observational and must never gate.