diff --git a/dag/gunbc/doc_graph_roots.dag b/dag/gunbc/doc_graph_roots.dag index c5b6e24c812..6db424f5d39 100644 --- a/dag/gunbc/doc_graph_roots.dag +++ b/dag/gunbc/doc_graph_roots.dag @@ -125,6 +125,17 @@ data body_lowering_production_consumer_ref: DeclarationRef = DeclarationRef { mo data mock_corpus_prefilter_control_note: String = "LIVE CONTROL FOR THE MOCK-CORPUS DECLARER OBSERVATION, and the reason this sentence names the type it does. This module imports v2 modules and sits under dag/, so it is exactly the shape that broke the floor when whole-tree published-mock declarer selection was decided by a substring scan: a documentary row naming PublishedMockCase made this file read as a corpus declarer, its closure could not resolve under the dag-only roots that precompute uses, and the run refused before any witness executed. The declarer decision is now the exact type-annotation check. This sentence therefore MENTIONS PublishedMockCase without declaring one, so if the substring shortcut is ever restored the floor reds here first." data hand_authored_doc_bind_incomings: List = [ + HandAuthoredDocBind { + home: PlanDoc, + slug: "fabric-ci-replacement", + primary_work: DeclarationRef { module_path: "gunbc.witness_floor_workflow", decl_name: "witness_floor_job", field: WholeDeclaration }, + additional_works: [ + DeclarationRef { module_path: "extdeps.github.checks", decl_name: "CheckRun", field: WholeDeclaration }, + DeclarationRef { module_path: "product.fabric.execution", decl_name: "ExecutionGrant", field: WholeDeclaration }, + DeclarationRef { module_path: "gunbc.ci_runner_target", decl_name: "CiRunnerTarget", field: WholeDeclaration }, + ], + dissolution: PlanRetiresWhen { condition: unbound_dissolution(description: "the cutover lands: the fabric issues the grant that runs the required floor, the check verdict is reported through the modeled Checks surface, and .github/workflows/witnesses.yml is deleted along with the emitter that produced it -- at which point this document describes a shipped system rather than a planned one and the row deletes with it. Registered at subject grain: the four systems bound here are the ones whose movement would make this document's purpose false. witness_floor_job is the thing being replaced and is bound deliberately, so that if it is renamed or restructured before the cutover this document reds rather than silently describing a job that no longer exists.") }, + }, HandAuthoredDocBind { home: PlanDoc, slug: "fabric-recut-program", diff --git a/dag/gunbc/fabric_witness_run.dag b/dag/gunbc/fabric_witness_run.dag new file mode 100644 index 00000000000..bf114c7e869 --- /dev/null +++ b/dag/gunbc/fabric_witness_run.dag @@ -0,0 +1,222 @@ +module gunbc.fabric_witness_run + +import std.types { NonEmptyStr } +import std.currency { CurrencyCode, Usd } +import std.nat { Nat } +import std.measure { HardwareThreadCount, hardware_thread_count, MoneyAmountMicro } +import product.fabric.identity { + FabricIdentity, WorkKey, ExecutionAttemptKey, OfferKey, GrantKey, + fabric_identity_eq, +} +import product.fabric.work { + Work, WorkContract, WorkStep, ResumabilityTier, ReceiptSatisfiable, + ExecutionRequirements, Shape, HardRequirements, + CapabilityManifestRef, TrustDomainRef, + ProgramRef, SourceManifestRef, ArtifactManifestRef, EffectContractRef, OutputContractRef, +} +import product.fabric.supply { Offer, OfferEligibility, offer_eligibility_for } +import product.fabric.demand { + Demand, AdmissionTerms, BuyTerms, SatisfactionRequirement, BudgetAccountId, ObservationReceiptRef, +} +import std.types { Timestamp } +import extdeps.exec.command { ArgvCommand } +import gunbc.ci_layer_roots { witness_layer_roots } +import std.types { String, List } + +// THE FLOOR'S PROGRAM AND CONTRACTS. NOTHING ELSE. +// +// This module used to own authorize_floor_run and a FloorRunRefusal coproduct. +// Both moved to product.fabric.supply, because comparing a Work's requirements +// against an Offer's capabilities is provider-neutral fungibility and nothing in +// ShapeNotCovered / TrustDomainMismatch / CapabilitiesNotOffered was ever about +// the floor. A gunbc module answering a fabric question is a gunbc module that +// will answer it differently from the next consumer. +// +// THE FABRIC RUNS THE WITNESSES. +// +// This is the execution half of the CI replacement: the argv that runs the +// required floor is PRODUCED BY THE FABRIC UNDER A GRANT, rather than written +// into a workflow step. No grant, no command -- so the authorization is not a +// label beside the run, it is the only thing that can yield a run. +// +// What is deliberately NOT here, because GitHub Actions still supplies it while +// it still triggers: polling, durable state, the compare-and-swap authority +// transition, supersession, and check publication. Those are the CONTROL plane +// and they are only needed once the fabric owns triggering. Attempting them now +// would build a second scheduler beside a live one. + +// The floor's contract. Every field is a fact about WHAT is promised, never +// about where or when it runs -- scheduling facts live in ExecutionRequirements +// beside it, so revising a thread estimate cannot mint a new semantic Work. +// +// Two steps, not one, because the workflow gates them separately and the v1 +// parse gate is a blocking step in its own right: collapsing them into one step +// would make a parse failure and a floor failure indistinguishable in the +// receipt, which is the distinction gunbc#8466 -> #8519 was paid to learn. + +fn floor_work_contract(source: SourceManifestRef) -> WorkContract { + WorkContract { + program: "gunbc.claim_executor", + steps: [ + WorkStep { step_id: "v1-dag-parse", resumability: ReceiptSatisfiable }, + WorkStep { step_id: "required-floor", resumability: ReceiptSatisfiable }, + ], + source: source, + inputs: floor_inputs_manifest_ref(), + runtime_closure: "gunbc.toolchain.rust-1.93.0+clippy+rustfmt", + effect_contract: "gunbc.floor.effects.read-only-corpus", + output_contract: "gunbc.floor.outputs.required-floor-summary", + } +} + +// OUR FLOOR IS arm64. THE CLASS IS NOT. +// +// This function previously spelled the architecture INSIDE a capability tag -- +// "gunbc.capabilities.linux-aarch64-rust-toolchain" -- which made an accident of +// our own supply look like a property of the execution class. Ampere is what we +// happen to stock; the product must serve x86 and arm at the lowest cost, and +// rented x86 is a supplier we now anticipate. A class that names its +// architecture cannot admit one it did not, which is the GitHub-vocabulary +// argument one layer over. +// +// The capability tag below therefore names the toolchain and NOT the machine. +// That our floor happens to be arm64 is a fact about our fleet's supply, not a +// property of the class -- see the note below for why it is not yet a value. +// +// WHY NO Architecture VALUE IS DECLARED HERE YET, and the receipt for it. +// Matching a declared architecture against an offer's requires the OFFER to +// carry one, and Offer has 25 constructors in tree, so the field and the arm +// that reads it land together. An earlier revision of this module declared +// `floor_architecture() -> Architecture = Aarch64` anyway, with a comment +// explaining that the consuming arm would come later. It had ZERO consumers, +// and v2.lens.inert_carrier caught it on CI as an unrostered inert carrier -- +// in the same commit whose message refused to add an unread field to Offer by +// citing review 53848's declared-but-nothing-derives-it defect. Refusing the +// defect on one side of a module boundary and authoring it on the other is the +// same defect, and the lens was a better reader of that than I was. +// +// So the architecture decision lives in prose and in the recut plan until it has +// a consumer. That is the honest state: a decision recorded is not a carrier +// modeled, and minting the carrier early buys nothing except a row that lies +// about being load-bearing. + +fn floor_execution_requirements() -> ExecutionRequirements { + ExecutionRequirements { + shape: Shape { hard: HardRequirements { threads: hardware_thread_count(count: 8) } }, + capabilities: "gunbc.capabilities.linux-rust-toolchain", + trust_domain: "gunbc.trust.internal-fleet", + } +} + +// THE SOURCE ROOTS ARE NOT THIS MODULE'S FACT. gunbc.ci_layer_roots +// witness_layer_roots is the existing authority -- the live workflow already +// folds it into its --source-root flags -- so declaring a list here was a fork +// of it, which is the violation one level above the one review 53848 caught. +// Consumed, not redeclared. +// +// The argv is a TOTAL fold over that authority, so adding a root changes the +// command with nothing to remember. An earlier revision of this module hand- +// expanded the pairs and declared a substrate gap with a dissolution trigger, +// on the grounds that expanding one element into two needs a flat-map and the +// substrate has neither flatten nor list destructuring. Both observations were +// true and the conclusion was wrong: `fold` exists and is exactly the primitive +// required -- the workflow module two files away was already using it for this +// same job. I declared a language-layer gap without enumerating the language, +// which is the failure my own notes name as searching by remembered name rather +// than reading the authority surface. + +fn floor_inputs_manifest_ref() -> ArtifactManifestRef { + concat("gunbc.floor.inputs.source-roots:", join(witness_layer_roots, "+")) +} + +// A modeled command, not a string blob. extdeps.exec.command already owns the +// argv carrier, so authoring a shell line here would have been a second +// representation of "how a process is invoked" -- the medium-as-string tell. + +fn floor_run_command() -> ArgvCommand { + ArgvCommand { + argv: fold( + witness_layer_roots, + init: ["target/release/claim_executor", "--required-floor"], + f: fn(acc, root) { append(acc, items: ["--source-root", root]) }, + ), + } +} + +// THE FLOOR IS AN ORDINARY PRICED DEMAND, NOT A PRIVILEGED PATH. +// +// gunbc's own CI enters the market on the same terms as any other demand, and +// this fold takes the budget ceiling as an argument precisely so that there is +// no arm here that admits a floor run without clearing a price. If self-CI +// bypassed the market, the opportunity cost of running our own work would not +// be a computable quantity and the arbitrage -- run our CI on our hardware, or +// sell that capacity and rent -- would degrade into a hand-maintained +// spreadsheet. +// +// Control-plane work is a separately privileged class. That exclusion must not +// leak here: a floor run is customer work that happens to be ours. + +fn floor_offer_eligibility

( + offer: Offer

, + maximum_buy_order: MoneyAmountMicro, + budget_currency: CurrencyCode, +) -> OfferEligibility { + offer_eligibility_for( + requirements: floor_execution_requirements(), + offer: offer, + maximum_buy_order: maximum_buy_order, + budget_currency: budget_currency, + ) +} + +// THE FLOOR'S DEMAND. This is where "ordinary priced demand" stops being a +// property of a fold's signature and becomes a row someone can read. +// +// REUSE IS A POLICY ON THE DEMAND, NOT A PROPERTY OF THE WORK. The cutover runs +// with terminal_receipt_may_satisfy = false and new_attempt_required = true, so +// an identical Work that already has an accepted receipt still executes. That +// is deliberately conservative -- it preserves today's Actions behaviour while +// the purity of the floor as a function of the tree is unproven -- and it is +// expressed HERE rather than in the Work key, because turning reuse on later +// must edit a policy row and not edit what a Work IS. The two were conflated in +// the plan's own section 13 and in the sign-off that answered it; the carriers +// never conflated them, which is why this needed no new modeling. +// +// step_reuse_permitted stays TRUE: a materialization provider may still +// accelerate a rerun. That is not the same claim as "the step completed" -- a +// build cache making the second attempt faster is not a receipt. + +fn floor_satisfaction_requirement() -> SatisfactionRequirement { + SatisfactionRequirement { + terminal_receipt_may_satisfy: false, + new_attempt_required: true, + step_reuse_permitted: true, + } +} + +fn floor_buy_terms( + account: BudgetAccountId, + reservation_price: MoneyAmountMicro, + maximum_buy_order: MoneyAmountMicro, +) -> BuyTerms { + BuyTerms { + budget_account: account, + currency: Usd, + reservation_price: reservation_price, + maximum_buy_order: maximum_buy_order, + } +} + +// priority_class is NOT a privilege escape. The floor competes on price like any +// other demand; a priority class orders demands that can all be afforded, it +// does not admit one that cannot. If this ever becomes the field that lets gunbc +// jump the queue, the arbitrage has been lost and the opportunity cost of our +// own work has stopped being computable. + +fn floor_admission_terms(buy: BuyTerms, deadline: Timestamp?) -> AdmissionTerms { + AdmissionTerms { + priority_class: "gunbc.ci.floor", + deadline: deadline, + buy: buy, + } +} diff --git a/dag/product/fabric/supply.dag b/dag/product/fabric/supply.dag index e3637f366e0..4d134ac28f1 100644 --- a/dag/product/fabric/supply.dag +++ b/dag/product/fabric/supply.dag @@ -1,11 +1,14 @@ module product.fabric.supply import std.types { Timestamp, Int } -import std.measure { MoneyAmountMicro, MoneyPerSecond, MoneyPerHour, MoneyOnce } +import std.measure { + MoneyAmountMicro, MoneyPerSecond, MoneyPerHour, MoneyOnce, + HardwareThreadCount, hardware_thread_count_value, money_amount_micro_count, +} import std.currency { CurrencyCode } import product.fabric.identity { FabricIdentity, OfferKey } import product.fabric.demand { ObservationReceiptRef } -import product.fabric.work { Shape, CapabilityManifestRef, TrustDomainRef } +import product.fabric.work { Shape, CapabilityManifestRef, TrustDomainRef, ExecutionRequirements } // An Offer is a supplier's statement of capacity it is willing to sell. Owned // capacity and rented capacity are ONE type: what differs is the evidence behind @@ -132,3 +135,98 @@ type Offer

{ type SelectionPolicy = | CheapestFungibleWithDelayValuation + +// FUNGIBILITY IS THE FABRIC'S QUESTION, NOT A CONSUMER'S. +// +// This fold arrived here from gunbc.fabric_witness_run, where it had been +// written as authorize_floor_run with a FloorRunRefusal coproduct. Nothing in +// it was ever about the floor: comparing a Work's requirements against an +// Offer's capabilities is provider-neutral, and work.dag's own header already +// calls fungibility "selection's first and load-bearing step". A consumer that +// owns this fold is a consumer that will answer it differently from the next +// consumer, which is the fork the fabric exists to prevent. +// +// AFFORDABILITY IS DELIBERATELY A SEPARATE ARM FROM CAPABILITY. A demand that +// cannot afford an offer and a demand an offer cannot serve are different +// facts with different remedies -- raise the buy order, or find another +// supplier -- and collapsing them into one "not eligible" loses which. + +type OfferEligibility + = OfferEligible + | ShapeNotCovered { required_threads: HardwareThreadCount, offered_threads: HardwareThreadCount } + | TrustDomainMismatch { required: TrustDomainRef, offered: TrustDomainRef } + | CapabilitiesNotOffered { required: CapabilityManifestRef, offered: CapabilityManifestRef } + | QuoteExceedsMaximumBuyOrder { quoted: MoneyAmountMicro, maximum: MoneyAmountMicro } + | QuoteCurrencyMismatch { quoted: CurrencyCode, budgeted: CurrencyCode } + | RateQuoteNotPriceableAgainstGrantCeiling { quoted: MoneyAmountMicro } + +fn offer_covers_shape(offered: Shape, needed: Shape) -> Bool { + hardware_thread_count_value(t: offered.hard.threads) >= hardware_thread_count_value(t: needed.hard.threads) +} + +// THIS FOLD SCREENS AFFORDABILITY. IT DOES NOT SELECT. +// +// It answers: can this demand pay this offer's asking price, in this currency, +// for a shape and trust domain that fit. Nothing here chooses among several +// eligible offers, and choosing is where the economics actually live. +// +// WHAT AN OFFER CARRIES, AND WHAT IT DOES NOT. The quote is a supplier-side +// ASKING PRICE for one grant. It is NOT an opportunity cost: what an hour could +// otherwise have earned depends on which OTHER demands could use it, so it is a +// property of the assignment evaluation and of the decision receipt, never a +// field on the supply row. Two earlier revisions of this note got this wrong in +// opposite directions -- first calling opportunity cost a demand-side question, +// then carrying it on the offer as a second cost -- and both are recorded here +// rather than reworded, because the sentence was cited downstream each time. +// +// Nor is marginal cash cost zero for an owned host: power and cooling are +// incurred by the act of running, so they are a real dispatch input. Only the +// HISTORICAL PURCHASE is sunk, and sunk cost is excluded from dispatch while +// being retained for accounting. +// +// WHAT SELECTION WOULD NEED, none of which exists here: the offer's availability +// interval, the supplier tariff with its billing quantum, rounding and minimum +// charge, the paid-through commitment state that makes an already-bought hour's +// marginal cash zero until it expires, transition and start costs, and the +// alternative uses that give an opportunity cost its value -- including the arm +// where there is NO feasible alternative use, so an hour about to expire idle is +// not priced as though a customer had been displaced, and the arm where the +// alternative is simply UNREAD, so unknown does not silently become zero. +// +// Until those land, this fold is honest as a screen and would be a lie as a +// selector. + +// The budget ceiling arrives as SCALARS rather than as demand.BuyTerms, because +// supply must not import demand -- a backward edge this module already carries +// once for ObservationReceiptRef and must not deepen. The demand side supplies +// them at the call site, which is also where they are known. + +fn offer_eligibility_for

( + requirements: ExecutionRequirements, + offer: Offer

, + maximum_buy_order: MoneyAmountMicro, + budget_currency: CurrencyCode, +) -> OfferEligibility { + if requirements.trust_domain != offer.trust_domain { + TrustDomainMismatch { required: requirements.trust_domain, offered: offer.trust_domain } + } else if requirements.capabilities != offer.capabilities { + CapabilitiesNotOffered { required: requirements.capabilities, offered: offer.capabilities } + } else if !offer_covers_shape(offered: offer.shape, needed: requirements.shape) { + ShapeNotCovered { + required_threads: requirements.shape.hard.threads, + offered_threads: offer.shape.hard.threads, + } + } else if offer_quote_currency(q: offer.quote) != budget_currency { + QuoteCurrencyMismatch { quoted: offer_quote_currency(q: offer.quote), budgeted: budget_currency } + } else { + match offer.quote { + QuotedPerSecond(r) => RateQuoteNotPriceableAgainstGrantCeiling { quoted: r.amount } + QuotedPerHour(r) => RateQuoteNotPriceableAgainstGrantCeiling { quoted: r.amount } + QuotedFlatPerGrant(r) => if money_amount_micro_count(m: r.amount) > money_amount_micro_count(m: maximum_buy_order) { + QuoteExceedsMaximumBuyOrder { quoted: r.amount, maximum: maximum_buy_order } + } else { + OfferEligible + } + } + } +} diff --git a/dag/test/claim/fabric_witness_run_test.dag b/dag/test/claim/fabric_witness_run_test.dag new file mode 100644 index 00000000000..da9fcacdea3 --- /dev/null +++ b/dag/test/claim/fabric_witness_run_test.dag @@ -0,0 +1,396 @@ +module test.claim.fabric_witness_run + +// The subject is ELIGIBILITY, so every witness here is a refusal or one of the +// controls that prove refusal is not the only reachable arm. A suite that only +// proved the happy path would establish that an offer can be accepted, which was +// never in doubt; what is in doubt is that it CANNOT be accepted when the +// executor does not satisfy the work, or when the demand cannot afford it. + +fn t_source() -> SourceManifestRef { + "gunbc.source.test-commit" +} + +fn t_max_buy() -> MoneyAmountMicro { + money_amount_micro(count: 1000000) +} + +fn t_offer( + threads: HardwareThreadCount, + caps: CapabilityManifestRef, + trust: TrustDomainRef, + quote: OfferQuote, +) -> Offer { + Offer { + id: FabricIdentity { principal: "gunbc", key: "offer-srv1" }, + executor: "gunbc", + shape: Shape { hard: HardRequirements { threads: threads } }, + capabilities: caps, + trust_domain: trust, + quantity_bound: 1, + ready_at: "2026-08-19T00:00:00Z", + quote: quote, + evidence: ObservedSupply { + observed_at: "2026-08-19T00:00:00Z", + receipt: "gunbc.observation.fleet-1", + }, + } +} + +fn t_quote(amount: MoneyAmountMicro, currency: CurrencyCode) -> OfferQuote { + QuotedFlatPerGrant(MoneyRate { amount: amount, currency: currency }) +} + +// THE OWNED-FLEET OFFER CARRIES AN ASKING PRICE, NOT AN OPPORTUNITY COST. +// +// An owned hour's marginal cash cost is not zero -- power and cooling are +// incurred by the act of running, so they are a real dispatch input -- and its +// OPPORTUNITY COST is not a field on the offer at all: what an hour could +// otherwise have earned depends on which OTHER demands could use it, so it +// belongs to the assignment evaluation, not to the supply row. An earlier +// revision of this comment claimed the quote WAS the opportunity cost; that was +// wrong in the same way the fold below is limited, and both are recorded rather +// than quietly reworded. +// +// What this scalar honestly is: the supplier-side ASKING PRICE for one grant. +// The fold compares it to what the demand will pay. That is AFFORDABILITY +// SCREENING and it is not selection -- nothing here chooses among several +// eligible offers, and choosing would need the interval, tariff, commitment +// state and alternative uses that no carrier holds yet. +fn t_fleet_offer() -> Offer { + t_offer( + threads: hardware_thread_count(count: 128), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: t_quote(amount: money_amount_micro(count: 40), currency: Usd), + ) +} + +fn eligible(e: OfferEligibility) -> Bool { + match e { + OfferEligible => true + ShapeNotCovered { required_threads: _, offered_threads: _ } => false + TrustDomainMismatch { required: _, offered: _ } => false + CapabilitiesNotOffered { required: _, offered: _ } => false + RateQuoteNotPriceableAgainstGrantCeiling { quoted: _ } => false + QuoteExceedsMaximumBuyOrder { quoted: _, maximum: _ } => false + QuoteCurrencyMismatch { quoted: _, budgeted: _ } => false + } +} + +fn floor_eligibility(offer: Offer) -> OfferEligibility { + floor_offer_eligibility(offer: offer, maximum_buy_order: t_max_buy(), budget_currency: Usd) +} + +// CONTROLS. Without these, every refusal below is satisfied by a fold that +// refuses unconditionally -- the shape a wall takes when it is actually a brick. + +test fn fleet_offer_is_eligible_for_the_floor() -> Bool { + eligible(e: floor_eligibility(offer: t_fleet_offer())) +} + +// The shape wall is an INEQUALITY. An offer larger than the requirement must +// still be eligible; one that rejected a bigger machine would be a bug wearing +// a wall's clothing. +test fn larger_offer_is_still_eligible() -> Bool { + eligible(e: floor_eligibility(offer: t_offer( + threads: hardware_thread_count(count: 256), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: t_quote(amount: money_amount_micro(count: 0), currency: Usd), + ))) +} + +// DISCRIMINATING REDS -- capability side. + +// A host outside the declared trust domain must be refused even when it is +// larger and cheaper: capacity does not substitute for trust. +test fn foreign_trust_domain_refuses_despite_ample_capacity() -> Bool { + match floor_eligibility(offer: t_offer( + threads: hardware_thread_count(count: 1024), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.public-cloud", + quote: t_quote(amount: money_amount_micro(count: 0), currency: Usd), + )) { + TrustDomainMismatch { required: _, offered: _ } => true + OfferEligible => false + ShapeNotCovered { required_threads: _, offered_threads: _ } => false + CapabilitiesNotOffered { required: _, offered: _ } => false + RateQuoteNotPriceableAgainstGrantCeiling { quoted: _ } => false + QuoteExceedsMaximumBuyOrder { quoted: _, maximum: _ } => false + QuoteCurrencyMismatch { quoted: _, budgeted: _ } => false + } +} + +test fn missing_capabilities_refuse() -> Bool { + !eligible(e: floor_eligibility(offer: t_offer( + threads: hardware_thread_count(count: 128), + caps: "gunbc.capabilities.linux-x86_64-no-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: t_quote(amount: money_amount_micro(count: 0), currency: Usd), + ))) +} + +test fn insufficient_threads_refuse() -> Bool { + !eligible(e: floor_eligibility(offer: t_offer( + threads: hardware_thread_count(count: 2), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: t_quote(amount: money_amount_micro(count: 0), currency: Usd), + ))) +} + +// DISCRIMINATING REDS -- price side. These are the arm that makes gunbc's own +// CI an ordinary priced demand rather than an always-admitted path. The fleet +// offer above quotes zero and still goes THROUGH this check; these prove the +// check is reachable and can refuse. + +test fn quote_above_the_maximum_buy_order_refuses() -> Bool { + match floor_eligibility(offer: t_offer( + threads: hardware_thread_count(count: 128), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: t_quote(amount: money_amount_micro(count: 9000000), currency: Usd), + )) { + QuoteExceedsMaximumBuyOrder { quoted: _, maximum: _ } => true + OfferEligible => false + ShapeNotCovered { required_threads: _, offered_threads: _ } => false + TrustDomainMismatch { required: _, offered: _ } => false + CapabilitiesNotOffered { required: _, offered: _ } => false + RateQuoteNotPriceableAgainstGrantCeiling { quoted: _ } => false + QuoteCurrencyMismatch { quoted: _, budgeted: _ } => false + } +} + +// Affordability and capability are SEPARATE arms with different remedies -- +// raise the buy order, or find another supplier -- so a currency mismatch must +// not be reported as an ordinary over-budget refusal. +test fn foreign_currency_quote_refuses_distinctly() -> Bool { + match floor_eligibility(offer: t_offer( + threads: hardware_thread_count(count: 128), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: t_quote(amount: money_amount_micro(count: 1), currency: Eur), + )) { + QuoteCurrencyMismatch { quoted: _, budgeted: _ } => true + OfferEligible => false + ShapeNotCovered { required_threads: _, offered_threads: _ } => false + TrustDomainMismatch { required: _, offered: _ } => false + CapabilitiesNotOffered { required: _, offered: _ } => false + RateQuoteNotPriceableAgainstGrantCeiling { quoted: _ } => false + QuoteExceedsMaximumBuyOrder { quoted: _, maximum: _ } => false + } +} + +// THE CEILING COMPARISON IS STRICT, AND THIS PINS THAT BOUNDARY. A zero-quoted +// offer clears a zero budget because zero is not GREATER than zero. +// +// This witness previously called t_fleet_offer() and asserted the same thing, +// with a comment reading "free capacity is affordable to a zero budget". That +// premise was true while the fleet fixture quoted zero and became FALSE the +// moment it was repriced -- 40 against a ceiling of 0 refuses -- so the witness +// went red on a claim nobody had reread. It now uses its OWN zero-quoted offer, +// because the property under test is the strictness of the comparison and has +// nothing to do with the fleet. +fn t_free_offer() -> Offer { + t_offer( + threads: hardware_thread_count(count: 128), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: t_quote(amount: money_amount_micro(count: 0), currency: Usd), + ) +} + +test fn a_zero_quote_clears_a_zero_ceiling_because_the_comparison_is_strict() -> Bool { + eligible(e: floor_offer_eligibility( + offer: t_free_offer(), + maximum_buy_order: money_amount_micro(count: 0), + budget_currency: Usd, + )) +} + +// The control that keeps the row above from being satisfied by a fold that +// admits everything: one micro over the ceiling refuses. +test fn one_micro_over_a_zero_ceiling_refuses() -> Bool { + !eligible(e: floor_offer_eligibility( + offer: t_offer( + threads: hardware_thread_count(count: 128), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: t_quote(amount: money_amount_micro(count: 1), currency: Usd), + ), + maximum_buy_order: money_amount_micro(count: 0), + budget_currency: Usd, + )) +} + +// THE COMMAND. Derived facts, not restatements. + +test fn floor_argv_carries_every_declared_source_root() -> Bool { + let argv = floor_run_command().argv + all(witness_layer_roots, r => any(argv, a => a == r)) + && length(xs: argv) == 2 + 2 * length(xs: witness_layer_roots) +} + +test fn fabric_argv_and_workflow_step_agree_on_source_roots() -> Bool { + let argv = floor_run_command().argv + all(witness_layer_roots, r => any(argv, a => a == r)) + && string_contains(s: witness_floor_run_script(), pattern: "--required-floor") + && all(witness_layer_roots, r => + string_contains(s: witness_floor_run_script(), pattern: concat("--source-root \"$ROOT/", concat(r, "\"")))) +} + +test fn floor_inputs_manifest_ref_folds_the_same_authority() -> Bool { + floor_inputs_manifest_ref() == "gunbc.floor.inputs.source-roots:dag+src/v2" +} + +// The floor and the v1 parse gate are TWO steps, so a receipt can say which one +// failed. That distinction is what gunbc#8466 -> #8519 paid to learn. +test fn floor_contract_keeps_the_parse_gate_as_its_own_step() -> Bool { + length(xs: floor_work_contract(source: t_source()).steps) == 2 +} + +// THE FLOOR IS AN ORDINARY PRICED DEMAND. These pin the two properties that, +// if they drift, silently turn gunbc's CI back into a privileged path -- which +// is invisible precisely because everything keeps working when it happens. + +// Reuse is disabled for the cutover, and it is disabled ON THE DEMAND. If this +// ever reads true while the floor's purity is unproven, an identical tree stops +// re-running and today's Actions behaviour has been changed by a policy edit +// nobody made deliberately. +test fn floor_demand_requires_a_new_attempt() -> Bool { + let s = floor_satisfaction_requirement() + s.new_attempt_required && !s.terminal_receipt_may_satisfy +} + +// Step reuse stays permitted: a materialization provider may accelerate a rerun. +// That is not the same claim as "the step completed", and the two must not +// collapse -- a build cache making the second attempt faster is not a receipt. +test fn floor_demand_still_permits_step_reuse() -> Bool { + floor_satisfaction_requirement().step_reuse_permitted +} + +// The buy terms carry a real ceiling that the eligibility fold reads. This is +// the join between "the floor is priced" and "the price is checked": a demand +// whose maximum buy order is below an offer's quote must not be eligible for it, +// and that must hold through the floor's own terms rather than only through the +// hand-built ones above. +test fn floor_buy_terms_ceiling_is_the_one_eligibility_reads() -> Bool { + let terms = floor_buy_terms( + account: "gunbc.budget.ci", + reservation_price: money_amount_micro(count: 0), + maximum_buy_order: money_amount_micro(count: 5), + ) + !eligible(e: floor_offer_eligibility( + offer: t_offer( + threads: hardware_thread_count(count: 128), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: t_quote(amount: money_amount_micro(count: 6), currency: Usd), + ), + maximum_buy_order: terms.maximum_buy_order, + budget_currency: terms.currency, + )) +} + +// priority_class must not become a privilege escape. This asserts the floor +// carries an ordinary class string and nothing structural distinguishes it from +// any other demand's -- there is no field here that could admit an unaffordable +// demand. +test fn floor_admission_carries_no_privilege_field() -> Bool { + let terms = floor_admission_terms( + buy: floor_buy_terms( + account: "gunbc.budget.ci", + reservation_price: money_amount_micro(count: 0), + maximum_buy_order: money_amount_micro(count: 5), + ), + deadline: none, + ) + terms.priority_class == "gunbc.ci.floor" +} + +// THE DISCRIMINATING PAIR FOR THE DIMENSIONAL DEFECT. maximum_buy_order is the +// ceiling on a single grant -- a TOTAL -- and an hourly rate is not one. Before +// the fix, the fold read .amount off the rate and compared it to the ceiling, so +// an hourly quote of 1 against a ceiling of 1000 was ADMITTED: the offer would +// bill 1 per hour for an unbounded number of hours, and nothing refused. This +// witness goes red against that fold and green only once the rate arms refuse. +test fn an_hourly_rate_is_not_priceable_against_a_per_grant_ceiling() -> Bool { + !eligible(e: floor_offer_eligibility( + offer: t_offer( + threads: hardware_thread_count(count: 128), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: QuotedPerHour(MoneyRate { amount: money_amount_micro(count: 1), currency: Usd }), + ), + maximum_buy_order: money_amount_micro(count: 1000), + budget_currency: Usd, + )) +} + +// The same amount quoted per SECOND and per HOUR differ by 3600x in fact and +// were indistinguishable to the old fold, which compared the bare scalar. Both +// must refuse for the same reason -- neither is a per-grant total -- so this +// asserts the erasure is gone rather than merely that one arm was special-cased. +test fn per_second_and_per_hour_quotes_both_refuse_rather_than_comparing_equal() -> Bool { + let per_second = floor_offer_eligibility( + offer: t_offer( + threads: hardware_thread_count(count: 128), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: QuotedPerSecond(MoneyRate { amount: money_amount_micro(count: 1), currency: Usd }), + ), + maximum_buy_order: money_amount_micro(count: 1000), + budget_currency: Usd, + ) + let per_hour = floor_offer_eligibility( + offer: t_offer( + threads: hardware_thread_count(count: 128), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: QuotedPerHour(MoneyRate { amount: money_amount_micro(count: 1), currency: Usd }), + ), + maximum_buy_order: money_amount_micro(count: 1000), + budget_currency: Usd, + ) + !eligible(e: per_second) && !eligible(e: per_hour) +} + +// THE POSITIVE CONTROL FOR THE REFUSAL. A per-grant total under the ceiling is +// still eligible, so the arm above refuses rates specifically and has not simply +// made every priced offer ineligible -- which would pass both reds above while +// destroying the fold. +test fn a_per_grant_total_under_the_ceiling_is_still_eligible() -> Bool { + eligible(e: floor_offer_eligibility( + offer: t_offer( + threads: hardware_thread_count(count: 128), + caps: "gunbc.capabilities.linux-rust-toolchain", + trust: "gunbc.trust.internal-fleet", + quote: t_quote(amount: money_amount_micro(count: 6), currency: Usd), + ), + maximum_buy_order: money_amount_micro(count: 1000), + budget_currency: Usd, + )) +} + +// SELF-CI IS NOT FREE BY CONSTRUCTION. The owned fleet carries an asking price, +// so a demand whose ceiling sits below it cannot claim the hour. This is the witness that would +// have been unwritable while the fleet quoted zero: at zero it clears every +// ceiling, so no ceiling can refuse it and the arm has no discriminating input. +test fn an_owned_hour_with_an_asking_price_can_be_outbid() -> Bool { + !eligible(e: floor_offer_eligibility( + offer: t_fleet_offer(), + maximum_buy_order: money_amount_micro(count: 39), + budget_currency: Usd, + )) +} + +// The control for the row above: the same owned-fleet offer IS eligible to a +// demand willing to pay its asking price. Without this, refusing at 39 is +// equally explained by the fleet offer having become ineligible outright. +test fn an_owned_hour_is_eligible_to_a_demand_that_meets_its_asking_price() -> Bool { + eligible(e: floor_offer_eligibility( + offer: t_fleet_offer(), + maximum_buy_order: money_amount_micro(count: 40), + budget_currency: Usd, + )) +} diff --git a/docs/plans/fabric-ci-replacement.md b/docs/plans/fabric-ci-replacement.md new file mode 100644 index 00000000000..298fc742d59 --- /dev/null +++ b/docs/plans/fabric-ci-replacement.md @@ -0,0 +1,1088 @@ +# Replacing GitHub Actions with the fabric: the plan + +**Status: APPROVED IN DIRECTION, amended. Sign-off received 2026-08-19 — "I approve the one-PR, +no-shadow replacement direction, but not the plan exactly as written" — with eight blocking +corrections and sixteen acceptance conditions, recorded in §11–§13 below. Earlier sections are kept +as authored, with their defects registered rather than edited away.** + +**Original framing:** Operator direction (2026-08-19): do the migration as one +PR, no long shadow process. That is the replacement doctrine's *default* — delete-first, one atomic +authority transition — rather than the gap-intolerant carve-out I had proposed. The plan below is +written to that ruling. + +## 1. What is actually being replaced + +GitHub Actions is a **fused authority**, and the recut program has just spent four cuts learning to +recognise one. It is simultaneously: + +``` +event source — a push or PR opened/synchronised +allocator — which machine, when, and how many at once +executor — checkout, toolchain, build, run +status sink — the red or green a human and the merge button read +log store — where the output goes +concurrency — cancel the superseded run +``` + +The fabric has authority over **allocation and execution**. It has no authority over webhooks or a +log UI, and pretending otherwise is how an MVP becomes a project. So this is the same move the side +chat ruled for `ComputeOffer` in §16 of the recut program: decompose the fused carrier, replace the +parts that have a home, and leave the rest cited rather than absorbed. + +## 2. The contract to preserve, measured from `.github/workflows/witnesses.yml` + +**The minimum replacement is not the smallest thing that runs the floor.** It must preserve every +refusal, or it has erased a correctness distinction rather than completed a migration. Measured +from the live workflow rather than remembered: + +| behaviour | current | must survive | +| --- | --- | --- | +| triggers | `workflow_dispatch`, push to `main`, PR to `main` (`opened`, `synchronize`, `reopened`, `ready_for_review`) | yes | +| runner | `[self-hosted, linux, arm64]` | yes — via the Cut C seam, not a literal | +| timeout | 180 minutes | yes, as a typed bound | +| concurrency | group per PR number or run id; **`cancel-in-progress` for PRs only** | yes — main runs never cancel | +| checkout | `fetch-depth: 0` (full history) | yes | +| toolchain | `setup-rust-toolchain@v1.16.0`, `cache: false` | yes | +| build | `claim_executor`, `gunbc`, `v1_src_dag_parse` | yes | +| gate 1 | `v1_src_dag_parse` — src/v1 `.dag` sources parse, **a separate step that must pass** | yes | +| gate 2 | `claim_executor --required-floor --source-root dag --source-root src/v2` | yes | +| env | `GUNBC_EXPECTED_RED_ROSTER_JOIN=expected_red_roster_join.tsv` | yes | +| failure | non-zero exit on any step is red | yes | + +Two of these are easy to lose silently and are called out for that reason: **the v1 parse gate is a +separate step** (it was invisible for 98 minutes once already, gunbc#8466 → #8519), and +**cancellation is asymmetric** — cancelling a main run would lose the only unconditional signal +that main is green. + +## 3. Why this is the right work, not a side quest + +`product.fabric.*` is 80 declarations across six modules with **no production consumer**. Every +guarantee it carries is currently type-level. This replacement makes the fabric load-bearing on the +one workload we already own end to end, which is the difference between a modeled market and a +market. + +It is also the **forcing function for Cut D**. Cut D stalled on two authorities that do not exist: +an execution class, and a fleet-derived runner authority (§17). Both are exactly what a CI +replacement must construct anyway — "what kind of machine does this run on" *is* an execution +class. So this work does not compete with the recut program; it grounds the part of it that had no +consumer to justify its shape. + +## 4. The terminal loop + +``` +poll GitHub extdeps.github.push_event / pulls + → Work "required floor at tree T", keyed by WorkContentKey + → Demand admission terms, budget, satisfaction requirement + → match eligible Offers fleet hosts, projected per Cut D's D2 + → ExecutionGrant accepted offer revision + reservation + executor + LeaseEpoch + → Attempt checkout, toolchain, build, v1 parse gate, required floor + → Receipt verdict, counts, evidence + → CheckRun PATCH extdeps.github.checks +``` + +**Polling, not webhooks**, deliberately: no inbound networking, no webhook secret, no listener to +secure. The fabric does not care how the event arrived, and a poller is strictly less +infrastructure than an endpoint. + +**The `LeaseEpoch` from Cut A is load-bearing here**, not decorative: two pollers, or a poller +restarted mid-run, must not both hold a grant on the same work. That is the fencing the epoch +exists for, and this is its first real consumer. + +## 5. What the one PR contains + +1. **`product.fabric` gains its executor** — the fold from Demand to Receipt. Types exist; the loop + does not. +2. **Fleet hosts project Offers** (Cut D's D2), with an explicit zero quote — owned supply. +3. **An execution class** for `linux/arm64 self-hosted`, which is what `RunnerSpec` should have + been derived from all along (§16), retiring Cut D's blocked precondition. +4. **The poller as a modeled systemd unit**, following `live_deploy_systemd_unit_for` — the only + genuinely new infrastructure, and the tree has the pattern. +5. **Check Run reporting** via the already-modeled `extdeps.github.checks` POST/PATCH. +6. **`.github/workflows/witnesses.yml` deleted**, and `gunbc.witness_floor_workflow` with it. + +## 6. Bootstrap and rollback, since there is no shadow + +**The hazard is real and must be stated rather than mitigated by optimism: after this merges, the +system that gates merges is the system that just changed.** If it is broken, the fix cannot be +merged through it. + +Three things make that acceptable rather than reckless: + +- The **cutover PR itself is validated by Actions**, because the PR is still under the old regime. + The last green Actions run is on the exact tree that contains the replacement. +- **Rollback is `git revert`**, and the operator merges manually today, so a revert can always be + landed by a human who can see the check is not reporting. This is the actual escape hatch, and it + is a person rather than a flag — which is the correct shape. +- The **fleet is already the execution substrate.** The runners are self-hosted on our hosts today, + so this changes the control plane, not the machines. + +## 7. What gets weaker — declared, not discovered + +**Independence of the control plane.** Today a fleet outage costs execution but GitHub still +queues the work and reports the status. Afterwards, a fleet outage means *nothing notices the +commit at all*. That is a genuine reduction and it belongs on the ladder as a declared rung, with +its trigger: it climbs when the poller has a liveness signal whose absence is itself observable — +a check that goes red when no poll has happened, not merely green when one has. + +**Log retention and the run UI** move from GitHub's storage to ours. The MVP keeps the Check Run +output as the human-readable surface; whole logs need a home before this is at parity, and naming +that gap is part of the plan rather than a follow-up someone discovers. + +**Concurrency semantics are re-implemented rather than inherited**, which is where a subtle +regression is most likely — specifically the asymmetry in §2. + +## 8. Open, for sign-off + +1. **Scope of the first cut: main pushes only, or PRs too?** PRs bring cancellation, merge-queue + semantics and per-PR concurrency. Main-only is dramatically smaller and still replaces a real + thing — but it leaves PR gating on Actions, which means the workflow file cannot be deleted, and + the doctrine's whole point is that the root goes in one motion. I do not think main-only is + coherent with the no-shadow ruling. I want that checked rather than assumed. +2. **Where does the poller run?** A fleet host is the obvious answer and the wrong one if that host + is also under test. Does the control plane need to be off-fleet to be honest? +3. **Is the execution class built here or in Cut D?** Building it here makes this PR bigger and + unblocks Cut D; building it in Cut D blocks this. I lean toward here, because a consumer-less + authority is what the recut program keeps refusing. +4. **What is the Work identity over?** The tree hash, so an identical tree does not re-run — which + would be a real gain over Actions, or a real hazard if the floor is not actually a pure function + of the tree. It reads pure; I have not proven it. + +## 9. Two defects in this plan, registered before the sign-off returns + +Both surfaced from the reviewer's working notes rather than the finished verdict, and both are +confirmed against the plan as written. Recorded now because one is an internal contradiction I +authored, and a plan that waits for permission to admit its own defect is doing the thing this +repository keeps charging elsewhere. + +### `workflow_dispatch` is listed as preserved and cannot be preserved by the design + +§2's table names `workflow_dispatch` among the triggers, marked **must survive: yes**. §4's loop +has **no manual trigger at all** — a poller notices commits; nobody can ask it for a run. So the +plan enumerates the contract correctly and then specifies something that cannot satisfy one row of +it. That is worse than omitting the row, because the table is what a reader would check the design +against. + +Deleting the workflow deletes the only manual re-run mechanism we have, and manual re-run is not a +convenience: it is how a human recovers from an infrastructure flake without pushing an empty +commit. The replacement needs an explicit *demand-creation* surface — which in fabric terms is the +honest shape anyway, since `workflow_dispatch` **is** a Demand authored by a person rather than by +an event. + +### Branch-head polling is a sampling, and sampling misses transitions + +The plan says "poll GitHub for new commit on main". Polling a branch *head* observes the current +value of a mutable pointer, so two pushes between polls collapse into one observation and the +intermediate commit never receives a check. That is not a rare race: it is the ordinary case for a +merge followed quickly by another merge, and it fails **silently** — the missed commit simply has +no check rather than a red one. + +This is the empty-observation narrow from DESIGN's failure-mode list, arriving through a different +door: "I sampled a pointer and saw one value" rendered as "there was one commit". The correct +construction is **reconciliation against durable state** — the set of commits that *should* carry a +verdict, differenced against the set that *do* — which is the shape the fleet spine already uses +(`Reconciliation`), not a poll-and-react loop. A reconciler that wakes up having +missed ten minutes converges; a poller that wakes up having missed ten minutes has lost the events. + +**Consequence for §4:** the loop's first arm is wrong as drawn. It is not +`poll → new commit → Work`; it is `desired check targets ⊖ observed check runs → Work per +difference`. The trigger becomes a wake-up rather than an event, and correctness stops depending +on the polling interval. + +### One framing to carry into the verdict + +The reviewer's phrase for the risk is worth keeping: whether polling preserves every refusal +**without becoming a new ambient authority**. A poller acts without being asked — nothing grants it +the right to spend the fleet on a commit. In the fabric's own vocabulary that is a Demand with no +authenticated author, which is precisely the shape `std.access` exists to refuse. The reconciler +framing improves this too: a reconciler derives its work from declared desired state, and declared +desired state has an author. + +## 10. Two blocking corrections — the plan tests the wrong commit, and the lease fences nothing + +Both from the reviewer's working notes ahead of the verdict. Both confirmed. Both are mine, and the +first is one I had already written down and failed to apply. + +### The PR subject is GitHub's synthetic merge commit, not the PR head + +`actions/checkout@v5` on a `pull_request` event checks out **`refs/pull/N/merge`** — a commit +GitHub *constructs* by merging the PR head into the base. So today's CI does not test the branch; +it tests **the branch as merged into main**. + +§4 of this plan says the Work is "required floor at tree T" derived from a polled commit. Applied +to PRs that would test the **head tree**, which is a different subject and a **strictly weaker +one**: a PR that is green in isolation and breaks when combined with main would pass. That is a +semantic-conflict class the current CI catches and the replacement as drawn would not — and losing +a refusal is exactly what §2 says disqualifies a minimum replacement. + +It is worse than an oversight, because I have this written down. My own note +(`branch-tree-identical-ci-subject-is-not`) opens: *"run A evaluated the synthetic merge of the +branch into one main commit, run B into a later main commit"* — and its stated rule is to check the +**merge subject, not just head**, before treating two runs as comparable. I applied it to comparing +runs and did not apply it to defining the Work. + +**Consequences the plan must now carry:** + +- The Work identity is **not** the head tree. It is the merge result, which means the identity + depends on **two** commits — PR head *and* base — so it changes when main moves even though the + PR did not. That also refutes the §8 question-4 hope that "an identical tree does not re-run": + the tree is not the input. +- Someone must **construct** the merge, since we would no longer be handed `refs/pull/N/merge`. + That is a real operation with a real failure mode (conflicts), and a conflicted merge is a typed + refusal, not an absent check. +- **`fetch-depth: 0` is load-bearing** and now visibly so: constructing a merge needs history. + +### `LeaseEpoch` alone fences nothing without a durable compare-and-swap + +§4 claims the epoch is load-bearing because "two pollers, or a poller restarted mid-run, must not +both hold a grant on the same work". The epoch is the right **token** and the sentence is +nonetheless wrong as an argument: a value does not exclude anyone. Two reconcilers can both read +epoch *N*, both believe they hold it, and both issue a grant. + +Fencing requires a **durable single-writer transition** — a compare-and-swap on persisted state +where exactly one writer observes success at each epoch, and the loser refuses rather than +proceeding. Without it, `LeaseEpoch` is a label on a race. + +This is the §5 trap in its purest form applied to my own design: I named a carrier and treated the +name as the guarantee, which is precisely what DESIGN §4b means by *richer type names are not +safety*. Cut A extracted the epoch as an immutable coordinate; **it never claimed to provide the +transition**, and I read the extraction as though it had. + +**So the plan acquires a prerequisite it did not have:** durable state with an atomic transition. +That is the first genuinely new *stateful* infrastructure in this proposal — the poller was new +process, this is new persistence — and it needs to be named as such rather than absorbed into "the +reconciler keeps track". Where that state lives, and what makes its CAS atomic, is now an open +question ahead of the four in §8. + +## 11. The merge gate is enforced by a ruleset, and the verdict's reading of it is wrong + +The sign-off states there is *"currently no GitHub-enforced required-check gate on `main`: branch +data reports zero required contexts and enforcement off"*, and builds a failure-mode argument on it. +**Measured directly, that is false**, and the mechanism of the error is one this repository has a +name for. + +``` +GET /repos/gunb-ai/gunbc/branches/main/protection → 403 Resource not accessible +GET /repos/gunb-ai/gunbc/rulesets → "passing CI", enforcement: ACTIVE +GET /repos/gunb-ai/gunbc/rulesets/16178731 → required_status_checks: [{ context: "witnesses" }] + plus deletion, non_fast_forward + conditions: ~DEFAULT_BRANCH +``` + +The gate exists, it is **active**, and it requires exactly the context `witnesses`. It is enforced +through a **ruleset**, not classic branch protection — a different API surface, which is why the +protection endpoint returns nothing useful. Reading "branch protection is empty" as "the branch is +unprotected" is the nearby-question failure: the endpoint answered a question adjacent to the one +asked, and answered it confidently. + +**Three consequences, and the first is a bootstrap trap that would have bitten during the cutover.** + +1. **Deleting `witnesses.yml` without addressing the ruleset makes every PR permanently + unmergeable** — including the rollback PR. Nothing would produce a check named `witnesses`, so + every PR sits at "Waiting for status to be reported" forever. The rollback path in §6 assumed a + human could merge a revert; under this ruleset a human cannot, without bypass authority that has + never been exercised. +2. **The good news is larger than the bad.** The required check carries **no `integration_id` pin** + — the parameter is `{"context": "witnesses"}` and nothing else. So *any* source publishing a + check named `witnesses` satisfies the rule. If the fabric's Check Run keeps that exact name, the + gate transition is a **no-op**: no ruleset edit, no window in which the gate is absent, and the + sign-off's concern about pinning the check to a GitHub App source becomes optional rather than + load-bearing. **Keeping the context name is therefore a design constraint, not a preference.** +3. The failure mode the sign-off wanted is **already the one we have**: poller unavailable → no new + check → PR unmergeable. That is fail-closed today, and the cutover must not weaken it. The + sign-off's premise that the merge button is currently not fail-closed on CI is what would have + licensed weakening it. + +`deletion` and `non_fast_forward` are also active on the default branch, so §6's rollback story +needs a named bypass authority that has actually been tested — the sign-off's condition 16, arrived +at from the opposite direction. + +## 12. The eight blocking corrections + +§9 and §10 already registered three of these; they are restated here in one place with the +sign-off's sharper form. + +1. **`LeaseEpoch` is not the serialization point.** It makes stale actuation decidable *after* a + canonical Grant exists; it does not stop two ticks from both reading free capacity and minting + different Grants against it. What is needed is a durable authority transition — *read canonical + generation N, consume idempotent observations, derive, commit N+1 iff N is still current* — plus + an **outbox** for GitHub mutations, because `CreateRun` succeeding and the process dying before + recording the returned id produces a duplicate Check Run. The modeled `CheckRun` accepts an + `external_id` on create but **does not retain it**, so there is no modeled exact join for + recovery. That join is part of the work. +2. **`extdeps.github.push_event` cannot poll.** It parses a webhook payload from + `GITHUB_EVENT_PATH`; once Actions is gone there is no such payload. Real polling operations are + needed. Main: persist a **SHA cursor**, enumerate every unseen descendant in ancestry order, + oldest first, and **refuse a non-descendant head as discontinuous history rather than silently + resetting the cursor**. Not timestamps — commit times are not a reliable cursor. +3. **PR Work uses the synthetic merge subject** (§10), and PR head identity and execution subject + stay **two separate facts** — the check correlates to the head, the run tests the merge. No + head-tree fallback: an unusable merge ref is a typed `MergeSubjectUnavailable` + (conflict / inaccessible / unobserved). A base-branch change that moves the merge subject + naturally creates a new Demand, which is a gain. +4. **`workflow_dispatch` does not survive workflow deletion** (§9) — GitHub only provides it while + the file exists on the default branch. It needs an explicit replacement creating a Demand with + *requested reexecution, exact ref, new Attempt required, prior receipt not sufficient*. A Check + Run "re-run" is not a substitute: re-requesting emits a webhook we would not receive. +5. **Check publication is a lifecycle, not a PATCH:** created `queued` when a Demand is admitted → + `in_progress` when the Attempt begins → `completed` with a conclusion, and `cancelled` / + `timed_out` as their own terminal arms. It needs a GitHub App identity with Checks write, whose + token stays on the control side and **never enters the execution workspace**. +6. **Logs and the semantic result need an owned store in this cut** — deferring it was refused + outright, and correctly: today's log surface cannot distinguish a complete semantic result from + a cancelled prefix from a reader-truncated prefix. Retained ordered artifact, terminal manifest + written last, manifest verifier, content-addressed blobs, `details_url` resolving to it. The + terminal conclusion derives from the verified artifact plus process termination plus the two + gate receipts — **never from scraping stdout**. +7. **Attempt precedes Grant.** My §4 loop had `Demand → Offer match → Grant → Attempt`, but the + `ExecutionGrant` carrier *names the Attempt it authorizes*, so that ordering cannot be + constructed. Attempt creation consumes no capacity; Grant issuance does. +8. **Source and toolchain materialization are real work**, not inherited. `FetchNoTags` does not + ground an attempt worktree in a named mirror. The Rust closure is pinned at `1.93.0` with + `clippy`/`rustfmt` and must be materialized and *verified*, not delegated to + `setup-rust-toolchain` — that action is the old realization, not a product concept. Preserve + Linux/AArch64 and the *reason* behind `cache: false` (no foreign post-job cleanup mutating a + cache shared with a live Attempt); do **not** preserve the literal `self-hosted` label, which is + a GitHub routing token whose subject disappears with Actions. `CARGO_TERM_COLOR=always` is in + the workflow and missing from §2's table — my omission. + +## 13. The four questions, answered + +1. **PRs too.** Main-only is coherent only as a partial transition that retains a workflow and two + execution authorities, contradicting the one-motion ruling — my reading was right. First cut + covers every unseen main commit, every current PR merge subject targeting main, and manual + reexecution. **No merge-queue semantics**: the current workflow has no `merge_group` trigger, so + it is not in the measured contract. PR concurrency is expressed as a **state-reconciliation + contract** — *every open PR targeting main has a terminal check for its current merge subject* — + rather than replaying `opened`/`reopened`/`ready_for_review`/`synchronize`. One behaviour change + is admitted deliberately: `ready_for_review` on an unchanged merge subject would no longer force + a re-run. Draft PRs currently *do* run CI (no draft exclusion) and that stays. +2. **Off-fleet.** The control-plane principal and store must not be an Offer the control plane + itself allocates; on-fleet placement lets one host failure remove observation, canonical state, + allocation and execution together. Minimum footprint: one small independent node, persistent + storage with snapshots, App key and token minting, a **one-shot tick** invoked by an external + cadence — not a resident interpreted loop — an outbox publisher, and point-to-point authority to + fleet executors. It runs no customer Work and never appears in the Offer roster. One node is a + declared single point of failure whose climb is an external observer or a second replica. +3. **Execution class lands here, narrowly.** Environment and entitlement for the floor — + Linux/AArch64, resource entitlement, source-materialization and toolchain-closure capability, + current trust profile. It must **not** contain a `RunnerSpec`, a `self-hosted` label, a public + SKU, a variance promise, or a physical host identity. The Offer-to-backing relation lands with + it, because selection can otherwise choose an Offer it cannot actuate. +4. **Not the tree hash** — my §8 hope is refused. `WorkContentKey` is over source subject + CI + contract digest + runtime closure + semantic environment + output contract, where source subject + is the exact **commit** identity (for PRs, the merge commit plus head/base correlation). Commit + rather than tree because the checkout materializes full history, so history may be a declared + input until proven otherwise; and reuse is **disabled for the cutover** — + `prior terminal receipt sufficient = no` — because witness budgets can depend on CPU and wall + measurements, witnesses may consume Git identity, and the output contract includes the + expected-red roster. **Dedup is not part of an authority cutover.** + +## 14. What "no shadow" means, and the one thing I need re-ruled + +The sign-off reads the operator's no-shadow instruction as **"no long-lived dual production +authority", not "no wet proof"**, and requires a bounded pre-merge canary: deploy the control plane +off-fleet, run the exact candidate merge subject through the fabric, publish a **non-required** +Check Run from the intended App, verify retained artifacts and teardown, disable intake, then merge +the one-motion cutover. + +I think that reading is right and the distinction is real — my §6 pre-merge argument was too thin, +because a green Actions run proves the tree resolves and its witnesses pass, and proves **nothing** +about token minting, Check Run creation, durable state recovery, grant acceptance, materialization +outside Actions, or artifact publication. + +**Operator ruling, 2026-08-19 — and it is better than either proposal:** *"wet proof before merge +can just come from the same PR that implements it — i.e. we make another job → confirm it works → +cutover/delete the first job."* + +So the wet proof is **a second job in the existing workflow**, not a hand-deployed canary. That +resolves condition 15 and improves on it in three ways the canary could not: + +- it is **automated and re-runs on every push to the PR**, so it cannot rot between the proof and + the merge — a manually deployed canary proves a moment, a job proves the current commit; +- it needs no separate pre-merge deployment step and no human remembering to disable intake; +- the evidence is a **green CI run on a commit in the PR's own history**, citable by SHA, rather + than a claim about something that happened on a control node. + +The sequence within the one PR: add job B exercising the fabric path alongside job A (`witnesses`) +→ push, both run, B green → same PR then deletes job A, the workflow, and the emitter. The final +diff has no workflow at all; the wet proof lives in the PR's history, not its head. + +**Verified as safe:** the required context is produced by the *job*, not the workflow — the live +check run on main is named `witnesses` from app `github-actions`, and the job id is `witnesses`. A +second job publishes its **own** check under its own name, which is not required and therefore +cannot disturb the gate. So the two-job phase is gate-neutral by construction. + +**The naming constraint from §11 becomes load-bearing here.** The context that gates is the job +named `witnesses`; deleting that job is precisely what removes the gate. Since the required check +carries no `integration_id` pin, the fabric's Check Run named `witnesses` satisfies the same rule +from a different source — so the gate never has a gap. **Job B must therefore NOT be named +`witnesses`** during the two-job phase (that name is still the live gate), and the fabric claims it +only at the cutover commit. + +## 15. The cutover PR cannot merge itself — the trap the two-job sequence walks into + +The reviewer's rollback analysis notes, as an inference, that a `pull_request` workflow executes +from the **PR's merge commit** rather than from the base branch — which is why a rollback PR that +*restores* `witnesses.yml` might run the restored workflow and satisfy its own required check. + +Run that inference in the other direction and it is not a convenience, it is a blocker: + +``` +the cutover PR DELETES .github/workflows/witnesses.yml + → the merge commit contains no such workflow + → Actions runs no job named `witnesses` for that PR + → the required context `witnesses` is never reported + → the ruleset holds the PR at "Waiting for status to be reported" + → THE CUTOVER PR CANNOT MERGE +``` + +The two-job sequence is correct right up to its final commit and then blocks on itself. Job B +proves the fabric wet, job A is deleted — and deleting job A is exactly what withdraws the check +the ruleset is waiting for. The same property that makes the two-job proof work (a PR's workflows +run from its own merge subject) is what makes the deletion self-blocking. + +**Three ways out, and they are not equally good.** + +1. **The fabric publishes `witnesses` for the cutover PR's own merge subject.** The control plane + is already deployed and observing by then — that is what the two-job phase established — so it + can satisfy the gate for the very PR that installs it. This is the elegant arm: the cutover is + gated by the system it is installing, which is the strongest possible wet proof, and the ruleset + never changes. It requires the fabric to be *authoritative* for one PR before merge, which is a + deliberate promotion of the canary rather than a shadow. +2. **Bypass.** Merge the cutover PR through a tested break-glass credential. This makes the + reviewer's bypass drill **mandatory rather than prudent**, and it means the cutover's success + depends on an authority we have never exercised. +3. **Two PRs** — one installs and proves the fabric while keeping job A, a second deletes job A + once the fabric is already publishing `witnesses`. This is the safest and it is in tension with + the one-motion ruling, though the *authority* still transitions in one motion; only the file + deletion is separated. + +**I have not tested the premise.** That a PR deleting a workflow causes it not to run for that PR +is documented GitHub behaviour and I am confident of it, but every confident structural claim in +this program that went untested today turned out wrong at least once, and this one decides the +shape of the final commit. The reviewer already proposed the machinery: a disposable branch with a +temporary active ruleset requiring a test context, and a PR that deletes the workflow producing it. +That test costs nothing and settles it. + +**It also sharpens the canary naming.** The canary must publish `fabric-witnesses-canary`, never +`witnesses`, so the old gate stays unambiguously load-bearing and it stays provable which +implementation satisfied it. Arm 1 above is the one moment that rule is deliberately suspended, and +it should be suspended once, knowingly, at the cutover commit — not by having both producers emit +the same context throughout the canary. + +## 16. Rollback, re-specified — and my claim about it was wrong + +I wrote in §11 that `deletion` and `non_fast_forward` mean "a human cannot merge a revert". That +overstated it. `non_fast_forward` prevents **force** pushes, not ordinary fast-forward updates, and +`deletion` prevents deleting the ref. **Neither bans a normal push to `main`.** The operative +blocker is the required check alone — which is a narrower and more actionable finding than the one +I recorded. + +What must actually be established before cutover: + +- **Effective rules for `main`**, from the branch-rules endpoint — it returns every active rule + applying to the branch including organisation-level ones, which neither classic branch protection + nor the repository ruleset list gives. +- **`bypass_actors` on the ruleset**, fetched with credentials that can see them — GitHub omits the + property from callers with insufficient ruleset access, so an empty reading is *not* evidence of + no bypass. (Exactly the shape of the mistake that produced the retracted claim.) +- Bypass **mode** per actor: `always`, `pull_request`, or `exempt`. A `pull_request`-mode actor + **cannot push directly** — it must open a PR and choose to bypass at merge. Being a repository + administrator is not itself sufficient. + +**Revised condition 16:** before cutover, at least one rollback route independent of the fabric is +executed end-to-end against an active equivalent ruleset, and a second independently shaped route +is prepared. Preferred pair: a rollback PR that restores and runs `witnesses.yml` as primary, and a +tested ruleset bypass by an independent credential as break-glass. The drill runs on a disposable +branch with its own temporary ruleset — negative control (ordinary identity refused), positive +control (break-glass succeeds and Rule Insights records a **Bypass**, an inspectable receipt rather +than a UI that looked like it would work) — then proves the drill branch's effective rules +equivalent to `main`'s. + +**The rollback credential, patch, and ruleset snapshot must not live only on the fabric control-plane +machine.** That is the whole point of a break-glass path. + +## 17. App pinning is deferred, deliberately + +Agreed both ways: **do not pin the App during the cutover.** No ruleset mutation inside an already +large authority transition, no interval where old and new required-check configuration disagree, +and the Actions-produced `witnesses` continues to gate the cutover PR. The cost is real but +**pre-existing**: an unpinned context can be published by anyone with sufficient repository +permission. + +Pinning is a later hardening cut, gated on the fabric App having produced `witnesses` on current +`main` and on every open PR's exact merge subject, its installation independently observed, the +rollback path not depending on that App being alive, and the current effective rules captured. +Then **update** the existing entry to add `integration_id` — never delete and recreate it. The +failure mode after pinning is not a protection gap but a fail-closed queue freeze if some open +subject lacks a fabric-authored check, which is why the pre-population is part of the cut. + +## 18. Product direction reframes this PR, and corrects four things in it + +From `swift-badger-524`, 2026-08-19. The reframe is not a relabelling: it changes what counts as +done, and two of the four corrections name defects in code already on this branch. + +### The reframe + +Not *"replace GitHub Actions with the fabric"* but: **build the first production CI binding, make +gunbc its first consumer, then cut gunbc's old executor authority over in one motion.** The test is +sharp — *you can successfully replace Actions and still have no saleable CI system.* The self-CI +cutover then wet-tests the exact path an external customer will use, which is a strictly stronger +proof than replacing our own workflow. + +The ownership split that follows: + +| layer | owns | +| --- | --- | +| `product.fabric` | provider-neutral negotiation, attempts, grants, receipts | +| GitHub binding | observations, subject correlation, demand commands, check projection | +| CI product layer | the public execution contract, isolation promise, billable receipt | +| `gunbc.*` | **only** the floor's program and contracts | + +**This locates a defect in what I already built.** `gunbc.fabric_witness_run.authorize_floor_run` +takes a `Work` and an `Offer` and answers a fungibility question — *does this executor satisfy these +requirements* — which is **provider-neutral fabric logic sitting in a gunbc module**. `FloorRunRefusal` +is likewise a general fungibility refusal wearing a floor-specific name: nothing in +`ShapeNotCovered`, `TrustDomainMismatch` or `CapabilitiesNotOffered` is about the floor. The gunbc +module should keep `floor_work_contract` and `floor_execution_requirements` and nothing else. + +### GitHub vocabulary must not leak into the fabric + +**Operator ruling: there is no "main vs PR" in the fabric — it is just how compute is negotiated.** +§13's answer to Q1 is right in substance and wrong in vocabulary: *"every unseen main commit, every +current PR merge subject, manual reexecution"* are three ways **this binding** obtains desired +demands, and they must never become three modes inside the fabric. A fabric that names its +provider's event kinds cannot admit a provider it did not anticipate — the same argument +`FabricIdentity` already makes for keeping the principal type open. + +### Self-CI is an ordinary priced demand + +**Confirmed as a gap in this branch:** `authorize_floor_run` checks shape, capabilities and trust +and **does not consider price at all**, so gunbc's floor is currently an always-admitted path. That +is precisely the privileged arm the arbitrage thesis forbids — if self-CI bypasses the market, the +opportunity cost of running our own work is not a computable quantity and the arbitrage degrades +into a hand-maintained spreadsheet. Cheap now, expensive to retrofit. + +Control-plane work is a separately privileged class, and that exclusion **must not leak to floor +runs**. + +### Placement: the invariant is not "off-fleet" + +§13's Q2 answer is superseded. The invariant is that **no single capacity, scheduler generation, +deployment, or failure domain may simultaneously remove observation, canonical state, allocation, +AND the only means of restoring them.** Meta-processes may run *on* the fabric, modeled as **control +work** — an execution class with reserved entitlement, its own trust domain, placement constraints +and receipts — distinct from customer work, which runs only from an admitted grant with no +control-plane credentials. + +Four prerequisites, not follow-ups: **(a)** the reserved placement class; **(b)** a durable +single-writer compare-and-swap — *`LeaseEpoch` is a fence, not a serialization mechanism, and +replication without CAS makes the race worse rather than better*; **(c)** bootstrap independence — +whatever starts the control plane must not require the control plane, so an external one-shot +cadence that consults nothing; **(d)** rollout durability — a new version takes over from a **dead** +predecessor, never requiring a graceful handoff. First honest rung: one management node plus an +independent watchdog plus a tested restore path, declared as an SPOF. + +### `WorkContentKey` bundles three concerns — split it + +§13's Q4 answer conflated **identity**, **reuse policy**, and **correlation**. Reuse-disabled-for- +cutover is a **policy over identities**, not a property of identity; per §3 policy is a workflow +fact. Turning reuse on later must not mean editing what things *are*. + +**The split is already available in the carriers** — verified rather than proposed: +`product.fabric.demand.SatisfactionRequirement { terminal_receipt_may_satisfy, new_attempt_required, +step_reuse_permitted }` lives on the **Demand**. So identity stays `WorkContentKey`, reuse policy +rides the Demand, and the cutover's "no reuse" is one Demand-side setting rather than a property +baked into what a Work *is*. + +## 19. The arbitrage runs within an architecture, not across it + +I asked for the arch-independent share of CI minutes, on the reasoning that it bounds the pool the +arbitrage operates over. **That framing was wrong and the number does not gate anything.** + +The arbitrage runs **between suppliers at a given architecture**: + +``` +arm64 work -> our Ampere fleet vs rented ARM (Hetzner CAX) +amd64 work -> rented x86 (CPX/CCX) vs any other x86 supplier +``` + +Every arm64 job — *including our entire floor, which is all arm64* — is already a live +own-vs-rented decision with two real offers. That is exactly the operator's arbitrage (run my CI on +my hardware, or sell that capacity and rent cheaper) and it requires **zero** architecture-independent +work to exist. + +Architecture-independent work is a **second-order bonus pool** on top: it additionally lets a job +cross arch lines to whichever is cheapest. Probably small — my guess there was likely right — but it +is not the mechanism, and I had promoted a second-order term to the load-bearing one. Under that +reading the pool looked nearly empty; under the correct one it is 100% of our floor plus 100% of +arm64 customer work, which is measurable from our own usage rather than from a market statistic +nobody publishes. + +**Open quantity, not a blocker:** the second-order cross-arch pool stays unmeasured. Not acquiring +it is deliberate — this is a revenue lane, not a research project. + +**What it changes for the build:** the two-offer condition that makes pricing live rather than +formal is satisfiable *today* with own-fleet plus one rented ARM offer. That is the next supply +increment after this slice, and it is what turns the price check from a formality into a market. + +## 20. The hourly minimum breaks the affordability fold I just built + +Price census finding (product direction, 2026-08-19): **Hetzner Cloud bills a one-hour minimum and +rounds every partial hour up.** An agent firing 200 five-minute jobs pays 200 hours and bills ~17. +So one-fresh-VM-per-job is not viable, and execution must **multiplex jobs onto long-lived hosts** +with per-job isolation from a microVM or container rather than a cloud-server lifecycle. + +Two things follow that are about this branch rather than about infrastructure. + +### `OfferQuote` cannot express the rule that changed the design + +``` +type OfferQuote = + | QuotedPerSecond(MoneyPerSecond) + | QuotedPerHour(MoneyPerHour) + | QuotedFlatPerGrant(MoneyOnce) +``` + +Every arm is a **rate and nothing else**. There is no minimum billable quantum, no rounding +direction, no minimum charge. So these three suppliers are indistinguishable in the model and +differ by ~12x in fact: + +| supplier | rate | rule the model cannot hold | +| --- | --- | --- | +| Hetzner Cloud | hourly | **1-hour minimum, partial hours round UP** | +| Ubicloud | 0.00125/min | per-minute, no stated minimum | +| Namespace | 0.002/min prepaid | 1-minute minimum, next 15s rounds **DOWN** | + +**The fabric already names this concept and does not carry it.** `execution.dag` states that the +supplier binding owns the billing rule — *"when billing starts, the quantum and rounding, minimum +charge, fixed and setup charges, caps, and when billing stops — because those differ per supplier"*. +That sentence is correct and there is no such carrier anywhere. The census turned a theoretical gap +into a load-bearing one. + +### The defect in `offer_eligibility_for` + +The affordability arm I landed compares `offer_quote_amount(q: offer.quote)` against the demand's +`maximum_buy_order`. **With a minimum billable quantum that is the wrong number.** For a five-minute +job against an hourly-minimum supplier the amount actually charged is the full hour, so the fold +would admit an offer the demand cannot afford — and it would do so silently, which is worse than +refusing: the refusal arm exists precisely so an unaffordable offer cannot be selected. + +The fold is not wrong for the own-fleet zero-quote case that exercises it today, which is exactly +why this is worth writing down now: **the first consumer does not exercise the defect**, and the +first consumer is the one whose shape hardens. + +**What affordability actually needs:** rate, quantum, minimum charge, rounding direction, and the +demand's expected duration — the last of which the `Demand` does not carry either. Until that +lands, the affordability arm is honest only for suppliers whose quantum is their rate's unit and +whose minimum is zero. + +### The asymmetry that is the business + +**We rent by the hour and sell by the minute.** That is not an accident to be normalised away — it +*is* the margin, and it means the fabric must express **two different billing rules on the two sides of +the same execution**: the supplier's rule on the offer we consume, and our rule on the offer we +publish. A model that assumes one billing rule per fabric cannot represent the arbitrage it exists +to run. + +Namespace's customer-facing rule is worth copying on our own side: minimum one minute, next 15 +seconds rounded **down**. Customer-favourable, nearly free, and legible. + +### The affordability arm is also dimensionally incoherent + +Verifying the paragraph above against the landed code found a second defect, independent of the +minimum quantum and simpler: + +``` +fn offer_quote_amount(q: OfferQuote) -> MoneyAmountMicro { + match q { + QuotedPerSecond(r) => r.amount // MoneyRate + QuotedPerHour(r) => r.amount // MoneyRate + QuotedFlatPerGrant(r) => r.amount // MoneyRate + } +} +``` + +It **strips the unit off a rate and returns it as a total**. The eligibility arm then compares that +against `maximum_buy_order: MoneyAmountMicro`, which `demand.dag` defines as *"the ceiling on a +single grant"* — a total. So the fold compares a rate to a total, and compares per-second and +per-hour quotes against the same ceiling **as if they were the same number**: an offer quoted per +hour and an offer quoted per second with identical `amount` fields are indistinguishable to the +arm, a 3600x error with no refusal. + +`std/measure.dag` had the distinction and the fold discarded it — `MoneyRate`, +`MoneyRate` and `MoneyRate` are three types, and `offer_quote_amount` erases the +parameter that separates them. This is worse than the quantum gap because the carrier already +exists: no new modeling is needed to refuse, only to stop erasing. It is the §5 tell exactly — +the arm is satisfiable while the realization lies. + +**FIXED on this branch.** The affordability arm now matches the quote: +`QuotedFlatPerGrant` is a per-grant total and compares as before, while the two rate arms refuse +with `RateQuoteNotPriceableAgainstGrantCeiling` carrying the amount they could not price. A +refusal, not a widen — being offered a rate we cannot yet price is now countable rather than +absorbed into `OfferEligible` — and it dissolves once the offer carries its billing rule and the +demand an expected duration, at which point pricing a rate is total. + +Evidence, executed in both directions rather than argued. Against the old comparison, with the +variant kept so the tests still resolve: + +| witness | old arm | new arm | +| --- | --- | --- | +| `an_hourly_rate_is_not_priceable_against_a_per_grant_ceiling` | `false` | `true` | +| `per_second_and_per_hour_quotes_both_refuse_rather_than_comparing_equal` | `false` | `true` | +| `a_per_grant_total_under_the_ceiling_is_still_eligible` | `true` | `true` | + +The positive control holding in **both** directions is what separates a specific refusal from a +fold broken into refusing everything — which would have passed both reds. + +**Found by verification, not by review.** Review 53896 approved this arm by name as *"split arms +with typed diagnostics — no absorbing fallback"* while it was comparing a rate to a total. It +surfaced only because a claim already sent upstream was checked against the landed code. + +### Acquisition is a third concept, not a case of placement + +Product direction, on this section: every arm of `OfferQuote` and the whole eligibility fold assume +an offer is **existing capacity we choose among**. Renting is not that — it **creates** capacity, +so the minimum billable quantum is also a **minimum acquisition commitment**. You cannot buy five +minutes of a Hetzner server; you buy an hour, speculatively, before knowing whether demand to fill +it arrives. + +So there are **three** decisions, not two: + +| concept | question | where the hourly minimum bites | +| --- | --- | --- | +| placement | which offer serves this demand | not here | +| pricing | what it costs and what we charge | quantum and rounding, per side | +| **acquisition / host lifecycle** | **when to rent, how long to hold, when to release** | **here** | + +Holding a rented host idle is pure loss; releasing it early wastes paid time; releasing and +re-acquiring inside one hour pays twice for the same hour. None of that is expressible today, and +it is the exact machinery that converts a negative spread at low occupancy into the margin at +saturation. + +**Recorded as a named third concept specifically so it is not absorbed into placement**, which is +where it would naturally and wrongly land — placement chooses among offers, acquisition +manufactures one at a cost floor of an hour. Not built in this slice. + +### What "correct enough" means here + +Per the operator's steer this is a make-money lane: the model must be correct enough to **bill +honestly**, not complete. That bounds the above to rate + quantum + rounding + minimum on both +sides, plus a named placeholder for acquisition. The dimensional defect was not in that bound — it +was a live wrong answer in landed code and cheap to fix, so it was fixed on its own terms rather +than scheduled. + +## 21. An offer carries two costs, and selection uses the second + +Product ruling (2026-08-20), which lands on the carrier this branch is shaping. + +| cost | what it is | when it is zero | +| --- | --- | --- | +| **marginal cash cost** | what spending this hour adds to the invoice | genuinely `0.00` for an owned hour, **and** for the remainder of a rented hour already bought | +| **opportunity cost** | what that hour could have been sold for | never zero for capacity anyone would pay for | + +**Selection reads opportunity cost.** Cash cost is a books-and-margin fact, not a placement input. +An owned hour entering selection at `0.00` outbids everything, so self-CI would always beat a +paying customer — the arbitrage defeated by the mechanism meant to run it. The cut applies +uniformly, which is the tell that it is right: a rented host inside its paid hour has the same +shape as an owned one, and nothing special-cases ownership. + +### The correction this makes to `supply.dag` + +The note landed earlier reads: *"the zero is a supply-side fact, the opportunity cost is a +demand-side question."* The second half is **wrong on homing**. Opportunity cost is what the hour +could have been sold for — a fact about the supply, carried on the offer as its second cost. The +note identified the right hazard (a zero quote is not an exemption from the price check) and put +the remedy in the wrong layer. Corrected rather than quietly reworded, because the sentence was +cited in a commit message and in a message upstream. + +### The first consumer encoded the forbidden shape + +`t_fleet_offer` quoted the owned fleet at **0** and `fleet_offer_is_eligible_for_the_floor` +asserted it clears — an owned hour entering selection at `0.00`, exactly what the ruling forbids. +The comment directly above it asserted that a zero quote *"is not an exemption from the price +check"* while the value handed to the check was zero, so it cleared every ceiling. **The comment +stated the principle the fixture violated.** + +Fixed on this branch: the fleet offer is quoted at its opportunity cost, and two witnesses pin the +consequence — `an_owned_hour_priced_at_opportunity_cost_can_be_outbid` (a demand below that price +cannot claim the hour) with its control `an_owned_hour_is_eligible_to_a_demand_that_meets_its_price` +(so the refusal is a price refusal, not the offer having become ineligible outright). **The first +of those was unwritable while the fleet quoted zero** — at zero it clears every ceiling, so no +ceiling can refuse it and the arm has no discriminating input. That is the sharpest statement of +why the fixture mattered: the wrong price did not merely misprice, it erased the test. + +`OfferQuote` remains a **single** scalar, so nothing yet *forces* it to be the opportunity cost +rather than the cash cost. That is the open carrier gap; pricing the fixture correctly keeps the +first consumer from hardening the wrong reading while the carrier is built. + +### Cost is a step function in time, not a scalar — and that is the multiplexing mechanism + +A rented host has **zero marginal cash cost for the remainder of its paid hour** and a full hour's +cost the instant it crosses the boundary. That converts the hourly minimum from a tax into a +**schedulable fact**: work landing inside an already-paid hour is free, so the fabric can prefer +it — and that preference *is* the multiplexing mechanism §20 called for, rather than a separate +policy bolted beside it. + +It is also why an offer needs a **lifetime**, not just a rate. So the billing carrier grows one +axis beyond §20: **rate + quantum + rounding + minimum, per side, plus a lifetime and a two-arm +cost.** This is the part that is expensive to retrofit and load-bearing for honest billing, so it +is in scope; the demand-derivative provisioning algorithm behind it is explicitly a later cut and +does not start. + +### Recorded for later, not now + +Provisioning as a rate-of-change problem — baseline demand served by cheap long-horizon supply, +volatile demand by elastic short-horizon — **grounded in baseload-vs-peaking from power markets** +rather than minting fresh vocabulary (§3: cite the real framework, do not re-coin it). And the +provisioning judgment removed into a once-ratified algorithm, whose corollary is that every +provisioning decision must emit a **receipt naming the inputs that drove it**, or "the fabric +decided" is unfalsifiable. + +## 22. The review tell, and the class underneath it + +### Review tell: a fixture at an absorbing extreme erases the arm + +Stated here rather than only in the commit that fixed it, because it is a **reviewer's question**, +not an incident. + +> A fixture priced at an absorbing extreme silently deletes the arm's discriminating power, and the +> suite still goes green because every remaining witness is a positive control. + +Zero was this instance; the class is any extreme that absorbs one side of a comparison — an +unbounded ceiling, an empty roster, a ⊤ budget, a deadline at infinity. The arm still executes, still +carries typed diagnostics, still *looks* exercised, and its discriminating half is dead **with +nothing reporting it, because a deleted test cannot fail.** + +**The question to ask of any refusal arm:** with the fixtures as they stand, can a witness be +written that makes this arm *fire*? If not, the arm is untested however many witnesses reference it. +`an_owned_hour_priced_at_opportunity_cost_can_be_outbid` was unwritable while the fleet quoted zero — +that is the sharpest form of the tell: the wrong fixture did not weaken the test, it made the test +inexpressible. + +### The class underneath: a richer-named carrier where a structural guarantee was needed + +Three defects on this branch are **one defect in three modules**, per the meta review: + +| instance | the name that carried | the guarantee that was needed | +| --- | --- | --- | +| `LeaseEpoch` as a fence (§10) | an epoch number | durable single-writer serialization | +| `FloorRunRefusal` homed in gunbc (§18) | a typed refusal | fungibility owned by the fabric | +| `offer_quote_amount` (§20) | `MoneyAmountMicro` | the unit that separates a rate from a total | + +DESIGN.md §4b already states the principle — *"richer type names are not safety; a brand, wrapper, +or `Validated` is cosmetic until construction and acceptance enforce the distinction."* What is +missing is not the rule but its **reader**. Per §3 and §6 the response to a third instance is to +promote the class, not to fix a fourth site in a fourth module. + +**And it is a measured review gap, not a suspicion.** Three times on this branch the prose stated the +principle correctly while the value violated it — architecture inside a capability string, a rate +compared as a total, a fixture at zero — and **review approved all three**. None is visible at the +name-and-shape layer, which is the layer structural review reads. §20 already records the sharpest +instance: review 53896 approved the affordability arm *by name* as *"split arms with typed +diagnostics — no absorbing fallback"* while it compared a rate to a total. More rounds of the same +review shape will not catch this class. + +**Next trigger:** the lens lands **between this PR and the cutover PR**, because durable +single-writer CAS is the exact next place the class can bite — a `CasToken` type name standing where +a durable compare-and-set was needed is the same defect with worse consequences: it fails as a +**lost update** rather than as a wrong number, and a lost update is the failure class that cannot be +detected after the fact. + +**Scoped to a decidable sub-class first, per §5's "never" trap.** *"A name promises more than the +construction delivers"* is **not decidable** — the promise lives in a reader's head, not in the +`Node` tree — so a lens scoped that way is an unbounded project that will not land between two PRs, +and the class sits unguarded while it is attempted. The general form stays a **reviewer's question**, +honestly and permanently, per the undecidable-residue rule. + +What is decidable, and the correction that matters: the tell above says an arm is *unfireable*, +which is **not** what any mechanism can measure. Whether a refusal arm *could* fire under some +unwritten fixture is undecidable; whether it *did* fire under the corpus we run is a measurement. +Those are different claims and only the second is checkable, so the lens must assert the second: + +> **no witness in the floor run ever constructed this refusal variant.** + +That is coproduct-variant observation coverage — decidable by execution, no intent inference, and it +catches the class from the **evidence side** rather than the naming side, which is where the class is +actually observable. It reports *untested*, never *unfireable*. + +**And that distinction is not a new caution — it is a failure mode DESIGN.md already names.** + +| | claim | about | +| --- | --- | --- | +| **unfireable** | no fixture could ever construct this | all *possible* fixtures | +| **untested** | no witness in the floor run constructed it | the corpus we *ran* | + +That is **⊥-as-answer conflated with ⊥-as-ignorance** — the *empty-observation narrow*, one level up +and pointed at our own evidence rather than at a diff. *"I did not observe it"* rendered as *"it +cannot happen"* is the same conflation as *"I could not compute what changed"* rendered as *"nothing +is affected"*, and it is strictly worse than the widen for the same reason: a widen is merely +expensive, a narrow is silently uncovered. So the lens says **untested** in those words, and +conflating them would re-introduce the undecidable claim inside the mechanism built to avoid it — +which is precisely the failure this scoping exists to prevent. + +Rejected as a candidate: **a declared vocabulary of guarantee words** (`epoch`, `fence`, `lease`, +`token`, `lock`, `guard`, `validated`) whose module declares no durable operation. It is decidable +and cheap, and it is **grep by another name** — §6 enforces with lenses, not greps — and it fires on +the naming layer we just established is exactly where the promise is *not* legible. An honest +`LeaseEpoch` and a lying one are spelled identically. + +**And the lens is the backstop, not the proof.** The CAS site needs a two-way executed control on +its own terms regardless of whether any lens lands — one run that loses an update against a +non-durable token, one that does not against the durable one. A lens that reports which arms went +unobserved cannot establish that the observed ones are correct. + +### Merge state, corrected + +An earlier report of *"two APPROVEs"* was wrong and is corrected here rather than left standing. +`dashboard-ops reviews 8576` is the source of truth: **8 approve verdicts, 1 distinct provider** +(`claude`), so `meets_two_approval_rule: false` at 1/2, with checks pending. Repeated approvals from +one provider do not compose, and the count of approve verdicts is not the readiness number. + +### Meta verdict + +`SHIP_WITH_DEBT` — merge this fold, cut the actual cutover as a separate PR, and land the +richer-name-as-guarantee lens between them. That matches §15's independent finding that the cutover +PR cannot merge itself, reached from the ruleset rather than from the review loop. + +## 23. The two-cost ruling corrected, and the mirror of the erased-test tell + +### The ruling's stronger half is withdrawn + +*"Selection uses opportunity cost; cash is a books fact, never a placement input"* is **wrong**, and +the Hetzner case refutes it: + +| | Hetzner paid-through slack | owned home host | opportunity alone | cash + opportunity | +| --- | --- | --- | --- | --- | +| **no competing demand** | cash 0, opp 0 | cash ~0.16 power/cooling, opp 0 | 0 vs 0 — a tie, cannot conclude | 0 vs 0.16 — takes the slack | +| **a paying customer wants the interval** | cash 0, opp = foregone contribution | cash 0.16, opp 0 | — | home wins, self-CI correctly displaced | + +The sum handles both cases; opportunity alone handles neither. The error was **conflating sunk cost +with marginal cash cost**: only the historical purchase is sunk, while power and cooling are +*incurred by the act of running* and so are real dispatch inputs. + +**Corrected objective:** marginal cash + opportunity + transition/start + valued delay and risk, +with historical purchase excluded from dispatch and retained for accounting. Grounded in the HM +Treasury Green Book (sunk costs must not affect the next decision; opportunity cost of already-paid +resources is assessed at next-best alternative use — **not** automatically a market list price and +**not** automatically nonzero). Short horizon is **economic dispatch**, long horizon is **unit +commitment**, and FERC fast-start rules already model commitment, startup, minimum-run and no-load +costs — the exact address for §21's baseload-vs-peaking grounding. + +### Two modeling corrections that supersede §21 + +1. **Opportunity cost is not a field on the offer.** It depends on which *other* demands could use + that offer, so it belongs to the **assignment evaluation** or the decision receipt. The offer + carries the facts costs *derive from*: availability interval, supplier tariff, billing quantum + and rounding, minimum charge, cap, paid-through commitment state, transition facts. Marginal + cash is not static either — it is a function of the requested interval against commitment state, + which is the staircase. +2. **The zero arm is essential.** `OpportunityCostStanding` needs **three** arms — + `OpportunityCostDerived`, `NoFeasibleAlternativeUse`, `OpportunityCostUnread` carrying an + obligation. Without the middle, an already-paid hour that will expire idle is priced as though an + imaginary customer had been displaced. Without the third, unknown silently becomes zero — the + state-space conflation exactly. + +### The mirror of the erased-test tell + +`zero_budget_still_clears_a_zero_quote` called `t_fleet_offer()` with `maximum_buy_order: 0` and +asserted eligible, under a comment reading *"free capacity is affordable to a zero budget"*. True +while the fleet quoted zero; **false the moment §21 repriced it** — 40 against a ceiling of 0 +refuses. Confirmed red by execution. + +So the class has two directions and only one was written down: + +| | what moves | what it does | +| --- | --- | --- | +| **erased test** (§22) | a fixture sits at an absorbing extreme | the arm becomes unfireable and nothing reports it | +| **stale premise** (here) | a fixture *changes* | a passing witness keeps asserting the old value's premise | + +**The review tell is different for each.** The first asks whether any witness can make the arm fire. +The second is a rule about editing: **after changing a fixture, re-read every witness naming the old +value in its name or its comment** — those are exactly the ones whose premise moved. `zero_budget…` +named the old value *in its own function name*. + +Repaired with its own `t_free_offer` fixture, since the property under test is the strictness of the +`>` comparison and never had anything to do with the fleet, plus +`one_micro_over_a_zero_ceiling_refuses` as the control. + +### Eligibility is not selection — the scope this PR actually holds + +The proof that no economic selector exists here is sharp: **rate quotes are refused for want of a +demand duration, and there is no fold choosing among multiple eligible offers on time-dependent +facts.** Representing marginal cash, supplier price and opportunity cost as **one scalar +distinguished only by comments** is the §22 class again — a richer name where a guarantee was +needed. + +So the economic language is **removed rather than defended**. `supply.dag` now states plainly that +the fold screens affordability and would be a lie as a selector, the quote is named an **asking +price**, and the two witnesses that named opportunity cost are renamed to match what they test. The +cost model lands in its own PR under the corrected objective above — which is the same narrow-next-PR +conclusion §22 reached from two other directions. + +## 24. The first completed floor run, and what it caught + +**Nineteen runs on this branch were cancelled by the next push before finishing; three failed; none +had ever completed.** The first one allowed to finish: + +``` +required-floor: planned=9747 executed=9747 terminal=9747 passed=9440 + known_red_held=306 failed=1 stale_quarantine=0 +required-floor: FAIL v2.test.lens_inert_carrier.inert_carrier_test.inert_carrier_no_unrostered_or_stale +``` + +Main was green at the same hour, so the failure is this branch's. Localised by running the lens's two +halves rather than guessing: `unrostered=0`, `stale=1` — a **rostered carrier had become live**. + +### It was `Offer`, and the row dissolved exactly as written + +`v2.lens.inert_carrier` rostered `Offer` with the reason *"modeled compute-fabric supply carrier +from the terminal-contract slice (#8413): declared ahead of the grant issuance that matches demand +against it, and **referenced only by `fabric_terminal_contract_witness_test` so far**"*, under the +dissolution condition *"a live consumer reads the carrier; the stale-roster check then forces this +row's deletion."* + +**This PR is that consumer** — `offer_eligibility_for` reads `offer.trust_domain`, +`.capabilities`, `.shape` and `.quote`. So the red is a **climb, not a regression**: a carrier left +the inert roster because something started reading it, and the lens refused to let the stale row +survive its own condition. The row is deleted; `unrostered=0`, `stale=0`, witness `true`. + +**This is the strongest evidence in the plan that the fabric acquired a real consumer**, and it is +better than any assertion the PR could make about itself: an independent lens, written before this +work, measured that `Offer` stopped being inert. + +### What the cancelled runs cost + +The stale-premise defect of §23 was live for four commits and the floor would have caught it. The +failure above was also live for the whole branch. Both were invisible because **a ~30-minute check +against a sub-30-minute push interval produces evidence at rate zero**, and nothing reports that — +cancelled runs do not look like failures and the PR reads *in progress* indefinitely. + +**Detection:** read the **conclusion histogram**, `gh run list --branch X --json conclusion`, not the +latest check. Successes of zero over nineteen attempts is the signal, and no surface shows it. + +## 25. Recorded and handed off, not fixed here + +While measuring the cost of §22's proposed lens, the same class turned up **inside the enforcement +layer**. Recorded precisely and deliberately left alone — it is not this lane's work, and the lane is +being narrowed for exactly this reason. + +`v2.lens.coverage` declares `CoverageDefectKey`, a 13-variant vocabulary of DESIGN's failure modes +including **`VacuousArm`** — §22's erased-test class, already named. Everything that consumes it is a +`data` row, a doc binding, and one witness whose entire body is +`coverage_defect_vacuous_arm != coverage_defect_parallel_authority`. **No fold anywhere produces a +list of *detected* defect keys from the corpus**; the module's generic `missing_coverage` helper is +consumed only by `v2.lens.mock_totality`, for an unrelated concern. + +**The receipt:** `v2.test.lens_coverage.near_miss_vacuous_not_parallel` declares +`near_miss_vacuous_node: Node` **and never reads it.** The input a detector would consume was +authored; the detector never was. Intent is normally unobservable — this is one of the rare cases +where an abandoned build leaves a receipt rather than an inference. + +**The method lesson, which generalises furthest:** *name plus green witness is the evidence shape +that reads as done.* To establish a class is covered, do not search for the **vocabulary** — search +for **the fold that produces a finding from the corpus**. A 13-variant enum with six hits and a +passing test is exactly what full coverage and zero coverage both look like from a grep. + +Not fixed here: the taxonomy exists but the detector and observation plumbing do not, and that is the +expensive half. Owner to be assigned outside this lane. diff --git a/src/v2/lens/inert_carrier.dag b/src/v2/lens/inert_carrier.dag index 20ebd9ad1b4..855f605bb0e 100644 --- a/src/v2/lens/inert_carrier.dag +++ b/src/v2/lens/inert_carrier.dag @@ -24,7 +24,6 @@ data inert_carrier_roster: List = [ inert_row(unit: "GitCliReportedVersion", reason: "modeled git-cli extdeps carrier with no live consumer yet"), inert_row(unit: "GitPushRequest", reason: "modeled git publication-transport carrier (extdeps.git.publication_transport slice 1): witness-tested in publication_publisher_witness_test with no non-test consumer yet; live transport binding awaits tools.publication_publisher slice"), inert_row(unit: "LocalAlias", reason: "inert G1 fixture type in qualified_module_projection_stripped_fixture.dag (see inert_carrier_local_alias_roster_note)"), - inert_row(unit: "Offer", reason: "modeled compute-fabric supply carrier from the terminal-contract slice (#8413): an executor's quoted/observed supply row, declared ahead of the grant issuance that matches demand against it, and referenced only by fabric_terminal_contract_witness_test so far"), inert_row(unit: "ReactHookSite", reason: "modeled typescript extdeps carrier with no live consumer yet"), inert_row(unit: "SystemdUnitStatus", reason: "modeled systemd extdeps carrier with no live consumer yet"), inert_row(unit: "TerminatingOneShotFrontier", reason: "modeled std.witness_purpose frontier coproduct from #7965 (Slice C-a): its variants name the two substrate absences that block a terminating one-shot classification, and only its own test file references it, so it is inert by the seed rule until the classifier consumes it"),