From b8c45d5c9636cf4d52311a3353e4799a94605cef Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 05:50:54 +0000 Subject: [PATCH 01/22] Provider-neutral model release authority; open-weight population cannot be narrowed by a distributor A probe of one distributor's registry returned 404 and was reported as "no local weights are published". That inference excluded three openly-published releases from the candidate population -- including the two highest-scoring open models available -- and nothing downstream could recover them, because excluding a candidate from the search is not visible as a ranking error. The defect was structural, not a lapse. `ModelArtifactProvenance` spells stock provenance as `PulledFromUpstream { publisher, reference: OllamaModelRef }`, so one distributor's vocabulary was the only way to say where a model came from and that distributor's silence had nowhere to land except as absence of the model. The same file already argued the principle and applied it correctly to the tuning arm, which carries a `ContentHash` rather than a reference precisely because "a lineage built on moving pointers records a story rather than a fact". gunbc.model.publication separates the four facts that were fused: release, weight publication, distribution channel, and packaging. Open-weight-ness is a CONSTRUCTOR of `ModelRelease`, so it is answered by matching the release and no channel observation participates -- there is no expressible path from a 404 to a change in that answer, rather than a check that would catch one. `ChannelPresence` has no arm spelling unqualified absence: `AbsentFromChannel` names the channel it is absent from, so "this model is unavailable" is not a sentence the type can produce. gunbc.model.population makes the narrowing invariant structural. A later stage is constructed from its parent by `filter`, so it cannot introduce a member the parent lacked, and refusing a stage cannot reach back into the parent because the parent is a separate immutable value the narrowing consumed. There is no writable state for the regression to be written into. `ReleaseDiscoverySource` carries a declared `CoverageScope`, and `open_weight_completeness` REFUSES when every source is a single distributor or an operator roster. The layer split alone does not repair this: the original error was a discovery error, and a census that only ever probes one distributor stays distributor-shaped while reporting that shape as completeness. Parameter counts are three sourced fields rather than one reconciled number. The two circulating totals for the fixture release differ by exactly the size of a separately-shipped speculative draft, so which one is meant decides whether a quantization fits a host. Evidence, established by execution: `w_all_population_claims_hold` returns true (eight claims -- four forbidden edges with the release that actually broke as the fixture, the discovery-coverage refusal, and two positive controls so the refusals discriminate). `w_red_arm_forbidden_edge_must_evaluate_false` returns false, asserting the forbidden edge directly; without it the suite would be a conjunction of things that happen to be true. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01HL72ndT9xg2dZEb6J8a2Gr --- dag/gunbc/model/choice.dag | 197 +++++++++++++++++ dag/gunbc/model/population.dag | 166 ++++++++++++++ dag/gunbc/model/publication.dag | 184 ++++++++++++++++ ...odel_population_narrowing_witness_test.dag | 203 ++++++++++++++++++ 4 files changed, 750 insertions(+) create mode 100644 dag/gunbc/model/choice.dag create mode 100644 dag/gunbc/model/population.dag create mode 100644 dag/gunbc/model/publication.dag create mode 100644 dag/test/claim/model/model_population_narrowing_witness_test.dag diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag new file mode 100644 index 00000000000..26bb5269998 --- /dev/null +++ b/dag/gunbc/model/choice.dag @@ -0,0 +1,197 @@ +module gunbc.model.choice + +import std.types { String, Bool, List, NonEmptyStr, Int } + +// CHOOSING WHAT TO SERVE, as a function rather than as an argument. +// +// WHY THIS IS A MODULE AND NOT A MEMO. The inputs to this decision move constantly: a new release +// appears, a quantization is published, a node is added, a context floor changes. Every time one +// moves, a prose recommendation is silently stale while still reading as authoritative -- and the +// people who must notice are the ones least able to, because the reasoning that produced it is gone. +// So the decision is computed from declared facts, and re-deriving it after a change costs a run +// rather than an argument. +// +// WHAT IT OPTIMIZES: the highest-quality artifact that FITS, subject to the operator's floors. Not +// the largest, not the fastest -- quality is the scarce thing when the goal is replacing a frontier +// model for programming, and every other axis is a constraint rather than an objective. +// +// WHAT IT REFUSES TO DO: pick silently when nothing fits. A refusal names the BINDING AXIS per +// candidate, which is what makes "would more hardware help" a derivable question instead of a +// debate. If every candidate is bound by memory, more nodes help; if they are bound by a context +// floor no published artifact reaches, more nodes buy nothing and the floor is the thing to move. + +// ============================ CAPACITY, at the three grains that actually differ ============================ + +// A vendor's nominal figure, the kernel's visible total, and a live availability reading are THREE +// DIFFERENT QUANTITIES, and confusing them has already produced a wrong fit decision. Nominal +// overstates by the firmware carveout; observed-available understates by whatever is resident at the +// moment of the reading. Only kernel_visible_bytes is a capacity, so only it is consumed here -- +// the others are retained so a reader can see the gap rather than rediscover it. +type NodeCapacity { + node_name: NonEmptyStr + nominal_bytes: Int + kernel_visible_bytes: Int + runtime_overhead_bytes: Int +} + +fn firmware_carveout_bytes(capacity: NodeCapacity) -> Int { + capacity.nominal_bytes - capacity.kernel_visible_bytes +} + +// The bytes a model may actually occupy: what the kernel can hand out, less what the serving runtime +// and OS need to stay alive. Deliberately NOT derived from a live availability sample. +fn allocatable_bytes(capacity: NodeCapacity) -> Int { + capacity.kernel_visible_bytes - capacity.runtime_overhead_bytes +} + +// ============================ CANDIDATES ============================ + +// Quality rank is an ORDINAL over quantizations of one release, not a score. Higher is better; the +// numbers carry no unit and must never be compared across releases, because a Q3 of a 284B model and +// a Q8 of a 30B model are not on one scale. Cross-release quality is the operator's subjective call, +// which is why it enters as a floor rather than as an objective. +type QuantizedCandidate { + model_name: NonEmptyStr + quant_label: NonEmptyStr + weight_bytes: Int + quality_rank: Int + declared_context: Int + kv_bytes_per_token: Int + measured_prefill_tokens_per_second: Int +} + +// KV scales with context and with how many sessions are simultaneously RESIDENT. Sessions parked to +// disk between turns do not count -- their KV is not in memory -- which is why hot_sessions is the +// parameter rather than a total session count. +fn kv_bytes(candidate: QuantizedCandidate, context_tokens: Int, hot_sessions: Int) -> Int { + candidate.kv_bytes_per_token * context_tokens * hot_sessions +} + +fn resident_bytes(candidate: QuantizedCandidate, context_tokens: Int, hot_sessions: Int) -> Int { + candidate.weight_bytes + kv_bytes(candidate: candidate, context_tokens: context_tokens, hot_sessions: hot_sessions) +} + +// ============================ CONSTRAINTS ============================ + +// The operator's floors. Each is a REFUSAL THRESHOLD, never a preference to be traded away silently: +// a candidate below a floor is not a worse choice, it is not a choice. +type ServingConstraints { + context_floor_tokens: Int + hot_sessions: Int + prefill_floor_tokens_per_second: Int +} + +// ============================ THE DECISION ============================ + +// Why a candidate was rejected, at the grain that tells you what to change. One axis per arm, so a +// census over rejections partitions cleanly into "buy hardware", "lower a floor", "wait for a +// better artifact". +type RejectionAxis + = DoesNotFitMemory { required_bytes: Int, allocatable_bytes: Int } + | ContextBelowFloor { declared: Int, floor: Int } + | PrefillBelowFloor { measured: Int, floor: Int } + +type CandidateVerdict + = Admissible { candidate: QuantizedCandidate, resident_bytes: Int, headroom_bytes: Int } + | Rejected { model_name: NonEmptyStr, quant_label: NonEmptyStr, axis: RejectionAxis } + +// Evaluated in a fixed order so the reported axis is deterministic: capability floors first, then +// fit. A candidate that fails several axes reports the first, and the ordering puts the axis the +// operator can act on soonest at the front. +fn evaluate_candidate( + candidate: QuantizedCandidate, + capacity: NodeCapacity, + constraints: ServingConstraints, +) -> CandidateVerdict { + let required = resident_bytes( + candidate: candidate, + context_tokens: constraints.context_floor_tokens, + hot_sessions: constraints.hot_sessions, + ) + let allocatable = allocatable_bytes(capacity: capacity) + match candidate.declared_context < constraints.context_floor_tokens { + true => Rejected { + model_name: candidate.model_name, + quant_label: candidate.quant_label, + axis: ContextBelowFloor { declared: candidate.declared_context, floor: constraints.context_floor_tokens }, + } + false => match candidate.measured_prefill_tokens_per_second < constraints.prefill_floor_tokens_per_second { + true => Rejected { + model_name: candidate.model_name, + quant_label: candidate.quant_label, + axis: PrefillBelowFloor { + measured: candidate.measured_prefill_tokens_per_second, + floor: constraints.prefill_floor_tokens_per_second, + }, + } + false => match required > allocatable { + true => Rejected { + model_name: candidate.model_name, + quant_label: candidate.quant_label, + axis: DoesNotFitMemory { required_bytes: required, allocatable_bytes: allocatable }, + } + false => Admissible { + candidate: candidate, + resident_bytes: required, + headroom_bytes: allocatable - required, + } + } + } + } +} + +fn verdict_is_admissible(verdict: CandidateVerdict) -> Bool { + match verdict { + Admissible { candidate: _, resident_bytes: _, headroom_bytes: _ } => true + Rejected { model_name: _, quant_label: _, axis: _ } => false + } +} + +fn verdict_quality(verdict: CandidateVerdict) -> Int { + match verdict { + Admissible { candidate: c, resident_bytes: _, headroom_bytes: _ } => c.quality_rank + Rejected { model_name: _, quant_label: _, axis: _ } => 0 - 1 + } +} + +// THE ANSWER, or a refusal carrying every rejection so the binding constraint is READ rather than +// argued. There is no arm that returns a best-effort pick when nothing fits: a degraded choice +// presented as the answer is the failure mode this whole exercise exists to remove. +type ServingChoice + = ChoseCandidate { verdict: CandidateVerdict } + | NoCandidateAdmissible { rejections: List } + +fn choose_serving_candidate( + candidates: List, + capacity: NodeCapacity, + constraints: ServingConstraints, +) -> ServingChoice { + let verdicts = map(candidates, c => + evaluate_candidate(candidate: c, capacity: capacity, constraints: constraints)) + let admissible = filter(verdicts, v => verdict_is_admissible(verdict: v)) + match length(admissible) == 0 { + true => NoCandidateAdmissible { rejections: verdicts } + false => ChoseCandidate { verdict: highest_quality(verdicts: admissible) } + } +} + +// Maximum by quality rank. Ties keep the earlier candidate, which makes the result stable under +// re-ordering of the input roster rather than dependent on it. +fn highest_quality(verdicts: List) -> CandidateVerdict { + fold(verdicts, head_of(verdicts), (best, v) => + match verdict_quality(verdict: v) > verdict_quality(verdict: best) { + true => v + false => best + }) +} + +fn head_of(verdicts: List) -> CandidateVerdict { + match first(verdicts) { + Present { value: v } => v + Absent => Rejected { + model_name: "none" as NonEmptyStr, + quant_label: "none" as NonEmptyStr, + axis: DoesNotFitMemory { required_bytes: 0, allocatable_bytes: 0 }, + } + } +} diff --git a/dag/gunbc/model/population.dag b/dag/gunbc/model/population.dag new file mode 100644 index 00000000000..87460d62613 --- /dev/null +++ b/dag/gunbc/model/population.dag @@ -0,0 +1,166 @@ +module gunbc.model.population + +import std.types { String, Bool, List, NonEmptyStr, Int, Timestamp } +import gunbc.model.publication { + ModelRelease, ReleaseIdentity, DistributionChannel, DistributionObservation, + release_identity, release_is_open_weight, distribution_channel_wire, +} + +// THE CANDIDATE POPULATION AND ITS NARROWING, built so that the invariant we care about is a +// property of the constructors rather than a rule someone remembers. +// +// THE INVARIANT: failure to enter a later population must never remove membership from an earlier +// one. Open weights exist regardless of whether a packager packaged them, a runtime loads them, a +// host fits them, or they answer well. Those are downstream verdicts about a REALIZATION; none is a +// fact about the release. +// +// HOW IT IS MADE STRUCTURAL RATHER THAN CHECKED. A narrowed population is not authored -- it is +// CONSTRUCTED FROM A PARENT by filtering, and it retains the parent. Members of a stage are a subset +// of its parent's members by the meaning of filter, so a stage cannot introduce a member the parent +// lacks, and refusing a stage cannot reach back into the parent because the parent is a separate +// immutable value that the narrowing consumed rather than modified. There is no writable state for a +// regression to be written into. A lens checking this afterwards would be a second representation of +// something the construction already guarantees. + +// The ordered stages. Each names what it is a verdict about, and every one below the first is a +// verdict about a REALIZATION of the release rather than about the release. +type PopulationStage + = OpenWeightStage + | LocallyPackagedStage + | RuntimeCompatibleStage + | SingleNodeFeasibleStage + | MultiNodeFeasibleStage + | ContextQualifiedStage + | CodingQualifiedStage + +fn population_stage_wire(stage: PopulationStage) -> String { + match stage { + OpenWeightStage => "open-weight" + LocallyPackagedStage => "locally-packaged" + RuntimeCompatibleStage => "runtime-compatible" + SingleNodeFeasibleStage => "single-node-feasible" + MultiNodeFeasibleStage => "multi-node-feasible" + ContextQualifiedStage => "context-qualified" + CodingQualifiedStage => "coding-qualified" + } +} + +// WHERE THE UNIVERSE CAME FROM, carried as data because the defect this repairs was a DISCOVERY +// error, not only a modeling one. +// +// Splitting release from distribution stops a probe's silence from meaning absence. It does NOT stop +// a census from being narrow because only one distributor was ever asked. If the source of the +// universe is implicit, the population is shaped by whichever prober happened to run, and it reports +// that shape as completeness. So the source is a modeled fact with a declared coverage scope, and a +// completeness question outside that scope REFUSES instead of answering from what it happens to hold. +type CoverageScope + = PublisherReleaseCatalog { publisher: NonEmptyStr } + | SingleDistributorCatalog { channel: DistributionChannel } + | OperatorAssertedRoster + +type ReleaseDiscoverySource { + authority: NonEmptyStr + coverage: CoverageScope + observed_at: Timestamp +} + +// A population, with its stage, its members, and the provenance that makes both auditable. +// +// narrowed_from is the parent stage where one exists. Its absence marks the root, and the root is the +// only population anyone authors directly. +type PopulationProvenance + = DiscoveredRoot { sources: List } + | NarrowedFrom { parent_stage: PopulationStage, rejected_count: Int } + +type ModelPopulation { + stage: PopulationStage + members: List + provenance: PopulationProvenance +} + +// THE ROOT. Open-weight membership is read from the release's CONSTRUCTOR, so no distribution +// observation participates in building it. A channel probe cannot shrink this set because a channel +// probe is not an input to it. +fn open_weight_population( + discovered: List, + sources: List, +) -> ModelPopulation { + ModelPopulation { + stage: OpenWeightStage, + members: filter(discovered, r => release_is_open_weight(subject: r)), + provenance: DiscoveredRoot { sources: sources }, + } +} + +// THE ONLY WAY TO BUILD A LATER STAGE. Members are a filter of the parent's members, so the subset +// property holds by construction rather than by assertion, and the count of what was rejected is +// retained so a narrowing that eliminates everything is visible rather than silent. +fn narrow_population( + parent: ModelPopulation, + stage: PopulationStage, + admits: List, +) -> ModelPopulation { + let kept = filter(parent.members, r => identity_in(roster: admits, candidate: release_identity(subject: r))) + ModelPopulation { + stage: stage, + members: kept, + provenance: NarrowedFrom { + parent_stage: parent.stage, + rejected_count: length(parent.members) - length(kept), + }, + } +} + +fn identity_in(roster: List, candidate: ReleaseIdentity) -> Bool { + length(filter(roster, i => + (i.publisher as String) == (candidate.publisher as String) + && (i.family as String) == (candidate.family as String) + && (i.revision as String) == (candidate.revision as String) + )) > 0 +} + +// =========================================================================================== +// COMPLETENESS, and its refusal. +// =========================================================================================== + +// A completeness question is answerable only when the sources behind the population actually cover +// the axis being asked about. Answering from a single distributor's catalog about the open-weight +// universe is the original defect in its general form: a real count over the wrong population. +type CompletenessVerdict + = CompletenessAnswerable { member_count: Int } + | CompletenessRefused { reason: String } + +fn open_weight_completeness(population: ModelPopulation) -> CompletenessVerdict { + match population.provenance { + NarrowedFrom { parent_stage: p, rejected_count: _ } => + CompletenessRefused { + reason: join([ + "completeness about the open-weight universe was asked of the ", + population_stage_wire(stage: population.stage), + " stage, which is a narrowing of ", population_stage_wire(stage: p), + " -- a later stage is a verdict about realizations and under-reports the universe", + ], ""), + } + DiscoveredRoot { sources: sources } => + match length(filter(sources, s => coverage_spans_open_weight_universe(coverage: s.coverage))) > 0 { + true => CompletenessAnswerable { member_count: length(population.members) } + false => CompletenessRefused { + reason: join([ + "no discovery source claims publisher-catalog coverage; every source is a single ", + "distributor or an operator roster, so an absent release is indistinguishable from ", + "one that was never enumerated", + ], ""), + } + } + } +} + +// A single distributor's catalog never spans the open-weight universe, and an operator roster is +// whatever a human remembered -- which is the other half of how the original census went stale. +fn coverage_spans_open_weight_universe(coverage: CoverageScope) -> Bool { + match coverage { + PublisherReleaseCatalog { publisher: _ } => true + SingleDistributorCatalog { channel: _ } => false + OperatorAssertedRoster => false + } +} diff --git a/dag/gunbc/model/publication.dag b/dag/gunbc/model/publication.dag new file mode 100644 index 00000000000..7f822954831 --- /dev/null +++ b/dag/gunbc/model/publication.dag @@ -0,0 +1,184 @@ +module gunbc.model.publication + +import std.types { String, Bool, List, NonEmptyStr, Int, Timestamp } +import std.content_hash { ContentHash, serialize_content_hash } + +// WHAT A MODEL RELEASE IS, modeled independently of anyone who serves, packages or hosts it. +// +// WHY THIS FILE EXISTS, stated as the defect it repairs rather than as a design preference. The +// predecessor authority spelled stock provenance as PulledFromUpstream { publisher, reference: +// OllamaModelRef }. That single field made one distributor's vocabulary the ONLY way to say where a +// model came from, and the consequence was not stylistic: a probe finding no Ollama manifest had no +// vocabulary in which to report "absent from this channel", so it reported absence of the model. The +// measured cost was that three releases whose weights are openly published -- one of them the +// highest-scoring open model available -- were recorded as nonexistent, and therefore never entered +// the population the whole program exists to choose from. Excluding the answer from the search is a +// worse failure than ranking it badly, because nothing downstream can recover it. +// +// So the repair is not a row asserting that a particular model exists. It is the removal of the edge +// that let a distributor's silence mean anything about a publisher's release. + +// A PUBLISHER'S RELEASE IDENTITY. Family and revision are separate because a family is what a +// human means and a revision is what the bytes are: DeepSeek-V4-Flash names a family across +// revisions, and DeepSeek-V4-Flash-0731 names one of them. Collapsing them would reintroduce a +// moving pointer as an identity, which is the defect the artifact layer already refuses. +type ModelFamilyName = NonEmptyStr where brand("ModelFamilyName") +type ModelRevisionName = NonEmptyStr where brand("ModelRevisionName") +type ModelPublisherName = NonEmptyStr where brand("ModelPublisherName") + +type ReleaseIdentity { + publisher: ModelPublisherName + family: ModelFamilyName + revision: ModelRevisionName +} + +// THE LICENSE UNDER WHICH WEIGHTS ARE PUBLISHED. Enumerated rather than a String because the +// question consumers ask -- may we run this on our own hardware -- is decidable from the license and +// must not be re-derived per consumer from prose. An unrecognized license is named, not defaulted: +// defaulting either way is a fabricated answer to a legal question. +type WeightLicense + = MitLicense + | Apache2License + | ModifiedMitLicense { steward: NonEmptyStr } + | NamedLicense { spdx_or_title: NonEmptyStr } + +// THE PARAMETER COUNTS, kept as THREE SOURCED FACTS rather than one reconciled number. +// +// This is not pedantry about a disagreement in the literature. A release that ships a speculative +// decoding draft as a SEPARATE artifact has two honest totals: the base network, and the base plus +// the attached draft. Both are true, they differ by the size of a real file, and which one is meant +// decides whether a given quantization fits a given host. Reconciling them into one number destroys +// the only signal that tells a reader which artifact set was measured. +// +// activated is the per-token count for a mixture-of-experts release and equals total for a dense +// one. It governs decode bandwidth; total governs whether the release is resident at all. +type ParameterCounts { + base_total: Int + activated_per_token: Int + attached_speculative: Int +} + +// WEIGHTS AS A PUBLISHED FACT: a repository, a revision, a license. This is what makes a release +// runnable by anyone, and it is a property of the PUBLISHER, not of any packager. +type WeightPublication { + repository: NonEmptyStr + revision: NonEmptyStr + license: WeightLicense + parameters: ParameterCounts +} + +// AN UPSTREAM CONTEXT CLAIM, deliberately typed as a CLAIM and never as a capability. +// +// A publisher declaring 1,048,576 positions via a scaling factor applied to a shorter trained window +// is stating what the architecture admits, not what the model retrieves. The gap between those is +// exactly where silent degradation lives: prefill accepts the tokens and the answer quietly stops +// depending on them. Nothing in this module promotes a declared number into a qualification, and the +// population that requires context is derived from OBSERVED retrieval elsewhere. +type ContextScaling + = NoScaling + | YarnScaling { original_positions: Int, factor: Int } + +type DeclaredContext { + max_positions: Int + scaling: ContextScaling +} + +// THE RELEASE ITSELF, and the load-bearing decision in this file: OPEN-WEIGHT-NESS IS A +// CONSTRUCTOR, not a field and not a predicate over distribution. +// +// Because the weight publication is carried by the OpenWeightRelease arm and by nothing else, the +// question "are the weights open" is answered by pattern-matching the release. No observation of any +// channel participates. There is therefore no expressible path from a distributor's 404 to a change +// in this answer -- not a check that would catch it, but no constructor that could write it. +type ModelRelease + = OpenWeightRelease { + identity: ReleaseIdentity + publication: WeightPublication + declared_context: DeclaredContext + } + | ClosedWeightRelease { + identity: ReleaseIdentity + declared_context: DeclaredContext + } + +fn release_identity(subject: ModelRelease) -> ReleaseIdentity { + match subject { + OpenWeightRelease { identity: i, publication: _, declared_context: _ } => i + ClosedWeightRelease { identity: i, declared_context: _ } => i + } +} + +fn release_is_open_weight(subject: ModelRelease) -> Bool { + match subject { + OpenWeightRelease { identity: _, publication: _, declared_context: _ } => true + ClosedWeightRelease { identity: _, declared_context: _ } => false + } +} + +// =========================================================================================== +// DISTRIBUTION: how some packager makes a release obtainable. One release, many channels, and a +// channel's answer is scoped to that channel by construction. +// =========================================================================================== + +// Each independently governed distributor is its own arm. Adding a distributor adds an arm; it never +// widens the meaning of the others, and it never touches ModelRelease. +type DistributionChannel + = PublisherRepository { host: NonEmptyStr } + | OllamaLibrary + | OllamaCloudEndpoint + | CommunityQuantRepository { host: NonEmptyStr, owner: NonEmptyStr } + +fn distribution_channel_wire(channel: DistributionChannel) -> String { + match channel { + PublisherRepository { host: h } => join(["publisher-repository:", h as String], "") + OllamaLibrary => "ollama-library" + OllamaCloudEndpoint => "ollama-cloud-endpoint" + CommunityQuantRepository { host: h, owner: o } => + join(["community-quant:", h as String, "/", o as String], "") + } +} + +// HOW A PACKAGED ARTIFACT IS SHAPED. Sharded packaging is called out because it is a REALIZATION +// constraint that has already refused a real install: a registry that cannot pull split files says +// nothing about the release and everything about that registry's loader. +type ArtifactPackaging + = SingleFileWeights { quantization: NonEmptyStr } + | ShardedWeights { quantization: NonEmptyStr, shard_count: Int } + | PublisherNativeCheckpoint + +// AN OBSERVATION ABOUT ONE CHANNEL, AND ONLY ONE CHANNEL. +// +// The absent arm names the channel it is absent FROM. There is no arm spelling unqualified absence, +// so "this model is unavailable" is not a sentence this type can produce. +type ChannelPresence + = PresentInChannel { reference: NonEmptyStr, packaging: ArtifactPackaging } + | AbsentFromChannel { probed_reference: NonEmptyStr } + +type DistributionObservation { + subject: ReleaseIdentity + channel: DistributionChannel + presence: ChannelPresence + observed_at: Timestamp +} + +fn observation_reports_presence(observation: DistributionObservation) -> Bool { + match observation.presence { + PresentInChannel { reference: _, packaging: _ } => true + AbsentFromChannel { probed_reference: _ } => false + } +} + +// The diagnostic a channel-absent observation is allowed to render. It states the channel, because a +// sentence that omits it is the exact sentence that caused the defect this module repairs. +fn channel_absence_diagnostic(observation: DistributionObservation) -> String { + match observation.presence { + PresentInChannel { reference: r, packaging: _ } => + join(["present in ", distribution_channel_wire(channel: observation.channel), " as ", r as String], "") + AbsentFromChannel { probed_reference: p } => + join([ + "not distributed by ", distribution_channel_wire(channel: observation.channel), + " under ", p as String, + " -- this states nothing about whether the release exists or its weights are open", + ], "") + } +} diff --git a/dag/test/claim/model/model_population_narrowing_witness_test.dag b/dag/test/claim/model/model_population_narrowing_witness_test.dag new file mode 100644 index 00000000000..26cfca470c0 --- /dev/null +++ b/dag/test/claim/model/model_population_narrowing_witness_test.dag @@ -0,0 +1,203 @@ +module test.claim.model.model_population_narrowing_witness_test + +import std.types { String, Bool, List, NonEmptyStr, Int, Timestamp } +import gunbc.model.publication { + ModelRelease, OpenWeightRelease, ClosedWeightRelease, + ReleaseIdentity, WeightPublication, ParameterCounts, DeclaredContext, + MitLicense, YarnScaling, NoScaling, + DistributionObservation, + OllamaLibrary, OllamaCloudEndpoint, + PresentInChannel, AbsentFromChannel, SingleFileWeights, + release_identity, release_is_open_weight, observation_reports_presence, +} +import gunbc.model.population { + ModelPopulation, + LocallyPackagedStage, SingleNodeFeasibleStage, + ReleaseDiscoverySource, + PublisherReleaseCatalog, SingleDistributorCatalog, + open_weight_population, narrow_population, + open_weight_completeness, CompletenessAnswerable, CompletenessRefused, +} + +// THE FIXTURE IS THE RELEASE THAT ACTUALLY BROKE. DeepSeek V4 Flash 0731 is openly published under +// MIT, is absent from the Ollama library, IS present as an Ollama cloud endpoint, and does not fit a +// single 118 GiB node at its published precision. Every one of those was read as "the model does not +// exist" by the predecessor authority, so it is the right discriminating subject: a fixture that was +// merely plausible would leave the original inference untested. +data deepseek_v4_flash_id: ReleaseIdentity = ReleaseIdentity { + publisher: "deepseek-ai" as NonEmptyStr, + family: "DeepSeek-V4-Flash" as NonEmptyStr, + revision: "0731" as NonEmptyStr, +} + +// Counts kept as three sourced facts. The attached figure is the DSpark speculative draft, which +// ships as its own file -- which is why the two circulating totals differ rather than one being wrong. +data deepseek_v4_flash: ModelRelease = OpenWeightRelease { + identity: deepseek_v4_flash_id, + publication: WeightPublication { + repository: "deepseek-ai/DeepSeek-V4-Flash-0731" as NonEmptyStr, + revision: "main" as NonEmptyStr, + license: MitLicense, + parameters: ParameterCounts { + base_total: 284, + activated_per_token: 13, + attached_speculative: 20, + }, + }, + declared_context: DeclaredContext { + max_positions: 1048576, + scaling: YarnScaling { original_positions: 65536, factor: 16 }, + }, +} + +data a_closed_release: ModelRelease = ClosedWeightRelease { + identity: ReleaseIdentity { + publisher: "example-lab" as NonEmptyStr, + family: "Proprietary-XL" as NonEmptyStr, + revision: "1" as NonEmptyStr, + }, + declared_context: DeclaredContext { max_positions: 200000, scaling: NoScaling }, +} + +data probed_at: Timestamp = 1788240000 as Timestamp + +data publisher_source: ReleaseDiscoverySource = ReleaseDiscoverySource { + authority: "huggingface publisher catalog" as NonEmptyStr, + coverage: PublisherReleaseCatalog { publisher: "deepseek-ai" as NonEmptyStr }, + observed_at: probed_at, +} + +data ollama_only_source: ReleaseDiscoverySource = ReleaseDiscoverySource { + authority: "ollama library index" as NonEmptyStr, + coverage: SingleDistributorCatalog { channel: OllamaLibrary }, + observed_at: probed_at, +} + +fn discovered() -> List { + [deepseek_v4_flash, a_closed_release] +} + +fn root() -> ModelPopulation { + open_weight_population(discovered: discovered(), sources: [publisher_source]) +} + +fn population_holds(population: ModelPopulation, id: ReleaseIdentity) -> Bool { + length(filter(population.members, r => + (release_identity(subject: r).family as String) == (id.family as String) + )) > 0 +} + +// ============================ THE FORBIDDEN EDGES ============================ +// Each is an inference that was actually drawn in production and was wrong. They are written as +// tests rather than as prose because a comment cannot go red. + +// RED 1 -- the original defect, verbatim. A 404 from one distributor's registry. +test fn w_ollama_local_absence_does_not_remove_from_the_open_weight_population() -> Bool { + let absent_from_ollama = DistributionObservation { + subject: deepseek_v4_flash_id, + channel: OllamaLibrary, + presence: AbsentFromChannel { probed_reference: "deepseek-v4-flash:latest" as NonEmptyStr }, + observed_at: probed_at, + } + !observation_reports_presence(observation: absent_from_ollama) + && release_is_open_weight(subject: deepseek_v4_flash) + && population_holds(population: root(), id: deepseek_v4_flash_id) +} + +// RED 2 -- a cloud tag existing was read as proof that local weights do not. +test fn w_ollama_cloud_presence_does_not_imply_local_weights_are_unavailable() -> Bool { + let cloud_present = DistributionObservation { + subject: deepseek_v4_flash_id, + channel: OllamaCloudEndpoint, + presence: PresentInChannel { + reference: "deepseek-v4-flash:cloud" as NonEmptyStr, + packaging: SingleFileWeights { quantization: "hosted" as NonEmptyStr }, + }, + observed_at: probed_at, + } + observation_reports_presence(observation: cloud_present) + && release_is_open_weight(subject: deepseek_v4_flash) + && population_holds(population: root(), id: deepseek_v4_flash_id) +} + +// RED 3 -- not fitting one host was read as the model being unavailable. +test fn w_single_node_infeasibility_does_not_remove_from_the_open_weight_population() -> Bool { + let feasible = narrow_population(parent: root(), stage: SingleNodeFeasibleStage, admits: []) + length(feasible.members) == 0 + && population_holds(population: root(), id: deepseek_v4_flash_id) +} + +// RED 4 -- narrowing must not introduce a member the parent lacked. +test fn w_narrowing_cannot_introduce_a_member_the_parent_lacked() -> Bool { + let stranger = ReleaseIdentity { + publisher: "nobody" as NonEmptyStr, + family: "Never-Discovered" as NonEmptyStr, + revision: "1" as NonEmptyStr, + } + let narrowed = narrow_population( + parent: root(), + stage: LocallyPackagedStage, + admits: [stranger, deepseek_v4_flash_id], + ) + length(narrowed.members) == 1 + && population_holds(population: narrowed, id: deepseek_v4_flash_id) +} + +// The positive control that the root filter discriminates at all -- without it every test above +// would pass on an unfiltered list. +test fn w_a_closed_weight_release_is_excluded_from_the_open_weight_population() -> Bool { + length(root().members) == 1 +} + +// RED 5 -- the DISCOVERY half, which the layer split alone does not repair. A census assembled from +// one distributor's catalog must refuse to say how complete it is. +test fn w_completeness_refuses_when_every_source_is_a_single_distributor() -> Bool { + let ollama_shaped = open_weight_population(discovered: discovered(), sources: [ollama_only_source]) + match open_weight_completeness(population: ollama_shaped) { + CompletenessRefused { reason: _ } => true + CompletenessAnswerable { member_count: _ } => false + } +} + +// Positive control for RED 5: publisher-catalog coverage DOES answer. +test fn w_completeness_answers_when_a_publisher_catalog_source_is_present() -> Bool { + match open_weight_completeness(population: root()) { + CompletenessAnswerable { member_count: n } => n == 1 + CompletenessRefused { reason: _ } => false + } +} + +// A later stage refuses completeness about the universe: it is a verdict about realizations and +// under-reports by construction. +test fn w_a_narrowed_stage_refuses_completeness_about_the_open_weight_universe() -> Bool { + let narrowed = narrow_population( + parent: root(), stage: LocallyPackagedStage, admits: [deepseek_v4_flash_id], + ) + match open_weight_completeness(population: narrowed) { + CompletenessRefused { reason: _ } => true + CompletenessAnswerable { member_count: _ } => false + } +} + +// AGGREGATE, so the whole carrier is established in ONE corpus resolve rather than eight. It is a +// conjunction of the individual claims and adds no coverage; it exists because each separate entry +// point re-resolves the corpus, and eight resolves exceed a remote dispatch budget that one fits +// inside. Every conjunct remains independently runnable, so a failure is still located. +test fn w_all_population_claims_hold() -> Bool { + w_ollama_local_absence_does_not_remove_from_the_open_weight_population() + && w_ollama_cloud_presence_does_not_imply_local_weights_are_unavailable() + && w_single_node_infeasibility_does_not_remove_from_the_open_weight_population() + && w_narrowing_cannot_introduce_a_member_the_parent_lacked() + && w_a_closed_weight_release_is_excluded_from_the_open_weight_population() + && w_completeness_refuses_when_every_source_is_a_single_distributor() + && w_completeness_answers_when_a_publisher_catalog_source_is_present() + && w_a_narrowed_stage_refuses_completeness_about_the_open_weight_universe() +} + +// THE RED ARM, kept enrolled. It asserts the FORBIDDEN edge -- that a channel-absent observation +// removes the release from the open-weight population -- and must therefore evaluate FALSE. Without +// it every claim above is a conjunction of things that happen to be true, and nothing demonstrates +// the carrier can fail. If this ever returns true, the structural separation has been undone. +test fn w_red_arm_forbidden_edge_must_evaluate_false() -> Bool { + !population_holds(population: root(), id: deepseek_v4_flash_id) +} From db4939fec7bf5a247f67e51d7fa62dc608bbf31d Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 05:52:54 +0000 Subject: [PATCH 02/22] Witness the serving selector against measured rows, with the flip control that makes it a selector choice.dag landed with no executed consumer, which is the specification-without- execution trap: it parsed, so the module index accepted it, and nothing ran it. The fixture is measured rather than plausible. Capacity is the kernel-visible MemTotal read from the running nodes (121.69 GiB), beside the vendor nominal (128 GiB) so the ~6.31 GiB firmware carveout is visible rather than rediscovered; quoting the nominal figure is what made a 128.1 GB build look feasible. KV is exact rather than estimated: the served model reports 43 blocks, ONE latent KV head, and key/value lengths of 512, so a token costs 88,064 bytes. The two candidates are the builds actually installed, at their real byte sizes. The claims are the operating conclusions, mechanised. At a 400k context floor the HIGHER-quality build is rejected on memory and the selector returns the 2-bit one -- precision and the context floor compete for one node's memory, and the function says so from the numbers instead of from an argument. The load-bearing test is the flip control: with the context floor dropped, the selector must choose the HIGHER-quality build. Without it every other claim is satisfied by a function that returns its first argument, and the suite would establish nothing about whether quality is maximized at all. The refusal arm is exercised too: with an unreachable floor the selector returns NoCandidateAdmissible carrying both rejections, so the binding axis is read rather than argued. A best-effort pick there would reintroduce exactly the silent degradation this module exists to remove. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01HL72ndT9xg2dZEb6J8a2Gr --- .../model/serving_choice_witness_test.dag | 151 ++++++++++++++++++ 1 file changed, 151 insertions(+) create mode 100644 dag/test/claim/model/serving_choice_witness_test.dag diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag new file mode 100644 index 00000000000..5aa4b91040f --- /dev/null +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -0,0 +1,151 @@ +module test.claim.model.serving_choice_witness_test + +import std.types { String, Bool, List, NonEmptyStr, Int } +import gunbc.model.choice { + NodeCapacity, QuantizedCandidate, ServingConstraints, + CandidateVerdict, Admissible, Rejected, + RejectionAxis, DoesNotFitMemory, ContextBelowFloor, PrefillBelowFloor, + ServingChoice, ChoseCandidate, NoCandidateAdmissible, + evaluate_candidate, choose_serving_candidate, + allocatable_bytes, firmware_carveout_bytes, kv_bytes, +} + +// THE FIXTURE IS MEASURED, not invented. Every number here was read off the running nodes or +// computed from the served model's own GGUF metadata, because a selector validated against +// plausible-looking inputs would agree with whatever the author already believed. +// +// kernel_visible 130,660,151,296 B = 121.69 GiB (/proc/meminfo MemTotal, both nodes) +// nominal 137,438,953,472 B = 128 GiB (vendor sheet) +// carveout 6,778,802,176 B = 6.31 GiB (firmware, never reaches the kernel) +// +// KV is exact rather than estimated: the architecture reports 43 blocks, ONE latent KV head (MLA), +// and key/value lengths of 512, so a token costs 43 * 1 * (512+512) * 2 = 88,064 bytes. +data node_capacity: NodeCapacity = NodeCapacity { + node_name: "spark-a3ee" as NonEmptyStr, + nominal_bytes: 137438953472, + kernel_visible_bytes: 130660151296, + runtime_overhead_bytes: 3000000000, +} + +data kv_per_token: Int = 88064 + +// The two builds actually installed. Quality rank is ordinal WITHIN this release: 3-bit outranks +// 2-bit. It says nothing about any other model, which is why cross-release quality never enters +// this function as a comparable number. +data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { + model_name: "DeepSeek-V4-Flash-0731" as NonEmptyStr, + quant_label: "IQ2_XXS" as NonEmptyStr, + weight_bytes: 86720111200, + quality_rank: 2, + declared_context: 1048576, + kv_bytes_per_token: kv_per_token, + measured_prefill_tokens_per_second: 197, +} + +data build_iq3_s: QuantizedCandidate = QuantizedCandidate { + model_name: "DeepSeek-V4-Flash-0731" as NonEmptyStr, + quant_label: "UD-IQ3_S" as NonEmptyStr, + weight_bytes: 116100000000, + quality_rank: 3, + declared_context: 1048576, + kv_bytes_per_token: kv_per_token, + measured_prefill_tokens_per_second: 197, +} + +fn installed() -> List { + [build_iq2_xxs, build_iq3_s] +} + +// The operator's stated floor: 400k of context, one hot session, no prefill floor asserted (that +// axis is a known cost rather than a refusal threshold today). +data floor_400k: ServingConstraints = ServingConstraints { + context_floor_tokens: 400000, + hot_sessions: 1, + prefill_floor_tokens_per_second: 0, +} + +data floor_8k: ServingConstraints = ServingConstraints { + context_floor_tokens: 8192, + hot_sessions: 1, + prefill_floor_tokens_per_second: 0, +} + +fn chosen_label(choice: ServingChoice) -> String { + match choice { + NoCandidateAdmissible { rejections: _ } => "none" + ChoseCandidate { verdict: v } => + match v { + Admissible { candidate: c, resident_bytes: _, headroom_bytes: _ } => c.quant_label as String + Rejected { model_name: _, quant_label: _, axis: _ } => "none" + } + } +} + +// ============================ THE CLAIMS ============================ + +// The three capacity grains are DIFFERENT, and only one of them is a capacity. Quoting the vendor +// figure is what made a 128.1 GB build look feasible. +test fn w_capacity_grains_differ_and_only_kernel_visible_is_spendable() -> Bool { + firmware_carveout_bytes(capacity: node_capacity) == 6778802176 + && allocatable_bytes(capacity: node_capacity) == 127660151296 + && allocatable_bytes(capacity: node_capacity) < node_capacity.nominal_bytes +} + +// KV at the operator's floor, from the model's own dimensions: 400,000 * 88,064 = 35.2 GB. +test fn w_kv_at_the_context_floor_is_derived_from_architecture() -> Bool { + kv_bytes(candidate: build_iq2_xxs, context_tokens: 400000, hot_sessions: 1) == 35225600000 +} + +// THE CLAIM THE MODULE EXISTS FOR. At a 400k floor the HIGHER-quality build is REJECTED ON MEMORY +// and the selector returns the 2-bit one. This is the answer that reversed the session's working +// assumption -- precision and the context floor compete for one node's memory, and the selector +// says so from the numbers rather than from an argument. +test fn w_at_the_400k_floor_the_higher_quality_build_is_rejected_on_memory() -> Bool { + match evaluate_candidate(candidate: build_iq3_s, capacity: node_capacity, constraints: floor_400k) { + Rejected { model_name: _, quant_label: _, axis: a } => + match a { + DoesNotFitMemory { required_bytes: _, allocatable_bytes: _ } => true + ContextBelowFloor { declared: _, floor: _ } => false + PrefillBelowFloor { measured: _, floor: _ } => false + } + Admissible { candidate: _, resident_bytes: _, headroom_bytes: _ } => false + } +} + +test fn w_at_the_400k_floor_the_two_bit_build_is_chosen() -> Bool { + chosen_label(choice: choose_serving_candidate( + candidates: installed(), capacity: node_capacity, constraints: floor_400k)) == "IQ2_XXS" +} + +// THE DISCRIMINATING CONTROL. Drop the context floor and the selector must flip to the HIGHER +// quality build -- proving it maximizes quality and is not simply always picking the smaller file. +// Without this, every claim above is satisfied by a function that returns the first candidate. +test fn w_without_the_context_floor_the_higher_quality_build_wins() -> Bool { + chosen_label(choice: choose_serving_candidate( + candidates: installed(), capacity: node_capacity, constraints: floor_8k)) == "UD-IQ3_S" +} + +// When nothing is admissible the selector REFUSES and hands back every rejection, so the binding +// axis is read rather than argued. A best-effort pick here would reintroduce the silent degradation +// the whole exercise removes. +test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { + let impossible = ServingConstraints { + context_floor_tokens: 8000000, + hot_sessions: 1, + prefill_floor_tokens_per_second: 0, + } + match choose_serving_candidate( + candidates: installed(), capacity: node_capacity, constraints: impossible) { + NoCandidateAdmissible { rejections: r } => length(r) == 2 + ChoseCandidate { verdict: _ } => false + } +} + +test fn w_all_serving_choice_claims_hold() -> Bool { + w_capacity_grains_differ_and_only_kernel_visible_is_spendable() + && w_kv_at_the_context_floor_is_derived_from_architecture() + && w_at_the_400k_floor_the_higher_quality_build_is_rejected_on_memory() + && w_at_the_400k_floor_the_two_bit_build_is_chosen() + && w_without_the_context_floor_the_higher_quality_build_wins() + && w_nothing_admissible_refuses_and_reports_every_rejection() +} From d25e38d58c04ed419a64c6f5603e903ee6fa685b Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 06:33:51 +0000 Subject: [PATCH 03/22] Route serving-choice quantities through std.measure; de-collide two corpus-wide variant names Two defects, both found by machinery rather than by reading. REVIEW FINDING (review 58079): every unit-bearing field in choice.dag was a bare Int -- capacities, weights, the KV rate, prefill throughput -- instead of consuming std/measure.dag carriers. Fixed at every site the review named plus the two it flagged as adjacent smells, since they are the same class: ByteSize for capacity/weights/KV, TokenCount for context, and a rate carrier for prefill. The verdict surface is converted too, so DoesNotFitMemory and the admissible arm no longer propagate flat scalars outward. quality_rank stays Int deliberately: it is a dimensionless ordinal, not a unit quantity. TokensPerSecond did not exist, so it is added to std.measure rather than minted locally -- consuming the single authority is the whole point of the finding. Its annotation records why it is a rate and not a TokenCount, and warns that prefill and decode rates share the type so the FIELD NAME has to separate them. The finding is sharper than a style rule and it bit this lane repeatedly: the session it came from produced four distinct wrong answers by attaching a correct number to the wrong population, including quoting a vendor-nominal memory figure as allocatable capacity, which made an over-capacity build look feasible. ByteSize and TokenCount make one of those classes unwritable instead of something a reader has to keep catching. CI FAILURE, same push: the required floor lane failed while the build lane passed. Cause was `Admissible` and `Rejected` as CandidateVerdict arms. Variant names resolve corpus-wide, so those two shadowed arms in four unrelated modules -- extdeps.tools.jq, extdeps.languages.markdown, extdeps.bmc.openbmc_fan_control and extdeps.provisioning.ubuntu_install_media_fetch -- none of which this change otherwise touches. That is why only the whole-corpus lane caught it, twenty minutes in. Renamed to ServingCandidateAdmissible / ServingCandidateRejected; CandidateRejected was rejected as a replacement because it collides in turn with the self-host door-observation modules. Verified by an exhaustive scan of all 38 variant arms introduced by these modules against the rest of the corpus, which now reports zero collisions. A local canary compile cannot establish this: compiling a victim module scopes to its own closure and never loads these modules, so only a whole-corpus pass observes the clash. Evidence after both fixes: selector aggregate true, selector flip control true (the control that distinguishes a quality-maximizing selector from one returning its first argument), population aggregate true, population red arm false. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01HL72ndT9xg2dZEb6J8a2Gr --- dag/gunbc/model/choice.dag | 99 ++++++++++--------- dag/std/measure.dag | 20 ++++ .../model/serving_choice_witness_test.dag | 66 +++++++------ 3 files changed, 110 insertions(+), 75 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index 26bb5269998..cfe8af92b29 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -1,6 +1,12 @@ module gunbc.model.choice import std.types { String, Bool, List, NonEmptyStr, Int } +import std.nat { Nat } +import std.measure { + ByteSize, byte_size, byte_size_count, + TokenCount, token_count, token_count_value, + TokensPerSecond, tokens_per_second_count, +} // CHOOSING WHAT TO SERVE, as a function rather than as an argument. // @@ -25,23 +31,23 @@ import std.types { String, Bool, List, NonEmptyStr, Int } // A vendor's nominal figure, the kernel's visible total, and a live availability reading are THREE // DIFFERENT QUANTITIES, and confusing them has already produced a wrong fit decision. Nominal // overstates by the firmware carveout; observed-available understates by whatever is resident at the -// moment of the reading. Only kernel_visible_bytes is a capacity, so only it is consumed here -- +// moment of the reading. Only kernel_visible is a capacity, so only it is consumed here -- // the others are retained so a reader can see the gap rather than rediscover it. type NodeCapacity { node_name: NonEmptyStr - nominal_bytes: Int - kernel_visible_bytes: Int - runtime_overhead_bytes: Int + nominal: ByteSize + kernel_visible: ByteSize + runtime_overhead: ByteSize } -fn firmware_carveout_bytes(capacity: NodeCapacity) -> Int { - capacity.nominal_bytes - capacity.kernel_visible_bytes +fn firmware_carveout(capacity: NodeCapacity) -> ByteSize { + byte_size(count: byte_size_count(b: capacity.nominal) - byte_size_count(b: capacity.kernel_visible)) } // The bytes a model may actually occupy: what the kernel can hand out, less what the serving runtime // and OS need to stay alive. Deliberately NOT derived from a live availability sample. -fn allocatable_bytes(capacity: NodeCapacity) -> Int { - capacity.kernel_visible_bytes - capacity.runtime_overhead_bytes +fn allocatable(capacity: NodeCapacity) -> ByteSize { + byte_size(count: byte_size_count(b: capacity.kernel_visible) - byte_size_count(b: capacity.runtime_overhead)) } // ============================ CANDIDATES ============================ @@ -53,22 +59,23 @@ fn allocatable_bytes(capacity: NodeCapacity) -> Int { type QuantizedCandidate { model_name: NonEmptyStr quant_label: NonEmptyStr - weight_bytes: Int + weights: ByteSize quality_rank: Int - declared_context: Int - kv_bytes_per_token: Int - measured_prefill_tokens_per_second: Int + declared_context: TokenCount + kv_per_token: ByteSize + measured_prefill_rate: TokensPerSecond } // KV scales with context and with how many sessions are simultaneously RESIDENT. Sessions parked to // disk between turns do not count -- their KV is not in memory -- which is why hot_sessions is the // parameter rather than a total session count. -fn kv_bytes(candidate: QuantizedCandidate, context_tokens: Int, hot_sessions: Int) -> Int { - candidate.kv_bytes_per_token * context_tokens * hot_sessions +fn kv_footprint(candidate: QuantizedCandidate, context: TokenCount, hot_sessions: Nat) -> ByteSize { + byte_size(count: byte_size_count(b: candidate.kv_per_token) * token_count_value(t: context) * hot_sessions) } -fn resident_bytes(candidate: QuantizedCandidate, context_tokens: Int, hot_sessions: Int) -> Int { - candidate.weight_bytes + kv_bytes(candidate: candidate, context_tokens: context_tokens, hot_sessions: hot_sessions) +fn resident_footprint(candidate: QuantizedCandidate, context: TokenCount, hot_sessions: Nat) -> ByteSize { + let kv = kv_footprint(candidate: candidate, context: context, hot_sessions: hot_sessions) + byte_size(count: byte_size_count(b: candidate.weights) + byte_size_count(b: kv)) } // ============================ CONSTRAINTS ============================ @@ -76,9 +83,9 @@ fn resident_bytes(candidate: QuantizedCandidate, context_tokens: Int, hot_sessio // The operator's floors. Each is a REFUSAL THRESHOLD, never a preference to be traded away silently: // a candidate below a floor is not a worse choice, it is not a choice. type ServingConstraints { - context_floor_tokens: Int - hot_sessions: Int - prefill_floor_tokens_per_second: Int + context_floor: TokenCount + hot_sessions: Nat + prefill_floor: TokensPerSecond } // ============================ THE DECISION ============================ @@ -87,13 +94,13 @@ type ServingConstraints { // census over rejections partitions cleanly into "buy hardware", "lower a floor", "wait for a // better artifact". type RejectionAxis - = DoesNotFitMemory { required_bytes: Int, allocatable_bytes: Int } - | ContextBelowFloor { declared: Int, floor: Int } - | PrefillBelowFloor { measured: Int, floor: Int } + = DoesNotFitMemory { required: ByteSize, allocatable: ByteSize } + | ContextBelowFloor { declared: TokenCount, floor: TokenCount } + | PrefillBelowFloor { measured: TokensPerSecond, floor: TokensPerSecond } type CandidateVerdict - = Admissible { candidate: QuantizedCandidate, resident_bytes: Int, headroom_bytes: Int } - | Rejected { model_name: NonEmptyStr, quant_label: NonEmptyStr, axis: RejectionAxis } + = ServingCandidateAdmissible { candidate: QuantizedCandidate, resident: ByteSize, headroom: ByteSize } + | ServingCandidateRejected { model_name: NonEmptyStr, quant_label: NonEmptyStr, axis: RejectionAxis } // Evaluated in a fixed order so the reported axis is deterministic: capability floors first, then // fit. A candidate that fails several axes reports the first, and the ordering puts the axis the @@ -103,37 +110,37 @@ fn evaluate_candidate( capacity: NodeCapacity, constraints: ServingConstraints, ) -> CandidateVerdict { - let required = resident_bytes( + let required = resident_footprint( candidate: candidate, - context_tokens: constraints.context_floor_tokens, + context: constraints.context_floor, hot_sessions: constraints.hot_sessions, ) - let allocatable = allocatable_bytes(capacity: capacity) - match candidate.declared_context < constraints.context_floor_tokens { - true => Rejected { + let room = allocatable(capacity: capacity) + match token_count_value(t: candidate.declared_context) < token_count_value(t: constraints.context_floor) { + true => ServingCandidateRejected { model_name: candidate.model_name, quant_label: candidate.quant_label, - axis: ContextBelowFloor { declared: candidate.declared_context, floor: constraints.context_floor_tokens }, + axis: ContextBelowFloor { declared: candidate.declared_context, floor: constraints.context_floor }, } - false => match candidate.measured_prefill_tokens_per_second < constraints.prefill_floor_tokens_per_second { - true => Rejected { + false => match tokens_per_second_count(r: candidate.measured_prefill_rate) < tokens_per_second_count(r: constraints.prefill_floor) { + true => ServingCandidateRejected { model_name: candidate.model_name, quant_label: candidate.quant_label, axis: PrefillBelowFloor { - measured: candidate.measured_prefill_tokens_per_second, - floor: constraints.prefill_floor_tokens_per_second, + measured: candidate.measured_prefill_rate, + floor: constraints.prefill_floor, }, } - false => match required > allocatable { - true => Rejected { + false => match byte_size_count(b: required) > byte_size_count(b: room) { + true => ServingCandidateRejected { model_name: candidate.model_name, quant_label: candidate.quant_label, - axis: DoesNotFitMemory { required_bytes: required, allocatable_bytes: allocatable }, + axis: DoesNotFitMemory { required: required, allocatable: room }, } - false => Admissible { + false => ServingCandidateAdmissible { candidate: candidate, - resident_bytes: required, - headroom_bytes: allocatable - required, + resident: required, + headroom: byte_size(count: byte_size_count(b: room) - byte_size_count(b: required)), } } } @@ -142,15 +149,15 @@ fn evaluate_candidate( fn verdict_is_admissible(verdict: CandidateVerdict) -> Bool { match verdict { - Admissible { candidate: _, resident_bytes: _, headroom_bytes: _ } => true - Rejected { model_name: _, quant_label: _, axis: _ } => false + ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => true + ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false } } fn verdict_quality(verdict: CandidateVerdict) -> Int { match verdict { - Admissible { candidate: c, resident_bytes: _, headroom_bytes: _ } => c.quality_rank - Rejected { model_name: _, quant_label: _, axis: _ } => 0 - 1 + ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => c.quality_rank + ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => 0 - 1 } } @@ -188,10 +195,10 @@ fn highest_quality(verdicts: List) -> CandidateVerdict { fn head_of(verdicts: List) -> CandidateVerdict { match first(verdicts) { Present { value: v } => v - Absent => Rejected { + Absent => ServingCandidateRejected { model_name: "none" as NonEmptyStr, quant_label: "none" as NonEmptyStr, - axis: DoesNotFitMemory { required_bytes: 0, allocatable_bytes: 0 }, + axis: DoesNotFitMemory { required: byte_size(count: 0), allocatable: byte_size(count: 0) }, } } } diff --git a/dag/std/measure.dag b/dag/std/measure.dag index 6b52e6e5c54..69becff91f8 100644 --- a/dag/std/measure.dag +++ b/dag/std/measure.dag @@ -891,6 +891,26 @@ fn token_count_value(t: TokenCount) -> Nat { measure_count(t) } +// A per-second rate of tokens -- the frequency family at One, period marker in the type, sibling of +// EventsPerMinute and MegatransfersPerSecond. It exists because token throughput is quoted two ways +// that are NOT the same quantity and get confused constantly: PREFILL rate (how fast an existing +// prompt is absorbed) and DECODE rate (how fast new tokens are emitted). Both are tokens per second, +// so the type cannot separate them -- the FIELD NAME must, and a consumer carrying only one of them +// is under-specified rather than merely terse. +// +// First consumer: gunbc.model.choice, where prefill rate is a serving-admission floor. It is a rate +// and not a TokenCount because a count answers "how many" and a rate answers "how fast"; assigning +// one to the other is the class this family exists to make unwritable. +type TokensPerSecond = Measure + +fn tokens_per_second(count: Nat) -> TokensPerSecond { + Measure { count: count } +} + +fn tokens_per_second_count(r: TokensPerSecond) -> Nat { + measure_count(r) +} + fn millicore(count: Nat) -> Millicore { Measure { count: count } } diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index 5aa4b91040f..4a06f1c65c5 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -1,13 +1,19 @@ module test.claim.model.serving_choice_witness_test import std.types { String, Bool, List, NonEmptyStr, Int } +import std.nat { Nat } +import std.measure { + ByteSize, byte_size, byte_size_count, + TokenCount, token_count, + TokensPerSecond, tokens_per_second, +} import gunbc.model.choice { NodeCapacity, QuantizedCandidate, ServingConstraints, - CandidateVerdict, Admissible, Rejected, + CandidateVerdict, ServingCandidateAdmissible, ServingCandidateRejected, RejectionAxis, DoesNotFitMemory, ContextBelowFloor, PrefillBelowFloor, ServingChoice, ChoseCandidate, NoCandidateAdmissible, evaluate_candidate, choose_serving_candidate, - allocatable_bytes, firmware_carveout_bytes, kv_bytes, + allocatable, firmware_carveout, kv_footprint, } // THE FIXTURE IS MEASURED, not invented. Every number here was read off the running nodes or @@ -22,12 +28,12 @@ import gunbc.model.choice { // and key/value lengths of 512, so a token costs 43 * 1 * (512+512) * 2 = 88,064 bytes. data node_capacity: NodeCapacity = NodeCapacity { node_name: "spark-a3ee" as NonEmptyStr, - nominal_bytes: 137438953472, - kernel_visible_bytes: 130660151296, - runtime_overhead_bytes: 3000000000, + nominal: byte_size(count: 137438953472), + kernel_visible: byte_size(count: 130660151296), + runtime_overhead: byte_size(count: 3000000000), } -data kv_per_token: Int = 88064 +data kv_per_token: ByteSize = byte_size(count: 88064) // The two builds actually installed. Quality rank is ordinal WITHIN this release: 3-bit outranks // 2-bit. It says nothing about any other model, which is why cross-release quality never enters @@ -35,21 +41,21 @@ data kv_per_token: Int = 88064 data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { model_name: "DeepSeek-V4-Flash-0731" as NonEmptyStr, quant_label: "IQ2_XXS" as NonEmptyStr, - weight_bytes: 86720111200, + weights: byte_size(count: 86720111200), quality_rank: 2, - declared_context: 1048576, - kv_bytes_per_token: kv_per_token, - measured_prefill_tokens_per_second: 197, + declared_context: token_count(count: 1048576), + kv_per_token: kv_per_token, + measured_prefill_rate: tokens_per_second(count: 197), } data build_iq3_s: QuantizedCandidate = QuantizedCandidate { model_name: "DeepSeek-V4-Flash-0731" as NonEmptyStr, quant_label: "UD-IQ3_S" as NonEmptyStr, - weight_bytes: 116100000000, + weights: byte_size(count: 116100000000), quality_rank: 3, - declared_context: 1048576, - kv_bytes_per_token: kv_per_token, - measured_prefill_tokens_per_second: 197, + declared_context: token_count(count: 1048576), + kv_per_token: kv_per_token, + measured_prefill_rate: tokens_per_second(count: 197), } fn installed() -> List { @@ -59,15 +65,15 @@ fn installed() -> List { // The operator's stated floor: 400k of context, one hot session, no prefill floor asserted (that // axis is a known cost rather than a refusal threshold today). data floor_400k: ServingConstraints = ServingConstraints { - context_floor_tokens: 400000, + context_floor: token_count(count: 400000), hot_sessions: 1, - prefill_floor_tokens_per_second: 0, + prefill_floor: tokens_per_second(count: 0), } data floor_8k: ServingConstraints = ServingConstraints { - context_floor_tokens: 8192, + context_floor: token_count(count: 8192), hot_sessions: 1, - prefill_floor_tokens_per_second: 0, + prefill_floor: tokens_per_second(count: 0), } fn chosen_label(choice: ServingChoice) -> String { @@ -75,8 +81,8 @@ fn chosen_label(choice: ServingChoice) -> String { NoCandidateAdmissible { rejections: _ } => "none" ChoseCandidate { verdict: v } => match v { - Admissible { candidate: c, resident_bytes: _, headroom_bytes: _ } => c.quant_label as String - Rejected { model_name: _, quant_label: _, axis: _ } => "none" + ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => c.quant_label as String + ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => "none" } } } @@ -86,14 +92,16 @@ fn chosen_label(choice: ServingChoice) -> String { // The three capacity grains are DIFFERENT, and only one of them is a capacity. Quoting the vendor // figure is what made a 128.1 GB build look feasible. test fn w_capacity_grains_differ_and_only_kernel_visible_is_spendable() -> Bool { - firmware_carveout_bytes(capacity: node_capacity) == 6778802176 - && allocatable_bytes(capacity: node_capacity) == 127660151296 - && allocatable_bytes(capacity: node_capacity) < node_capacity.nominal_bytes + byte_size_count(b: firmware_carveout(capacity: node_capacity)) == 6778802176 + && byte_size_count(b: allocatable(capacity: node_capacity)) == 127660151296 + && byte_size_count(b: allocatable(capacity: node_capacity)) + < byte_size_count(b: node_capacity.nominal) } // KV at the operator's floor, from the model's own dimensions: 400,000 * 88,064 = 35.2 GB. test fn w_kv_at_the_context_floor_is_derived_from_architecture() -> Bool { - kv_bytes(candidate: build_iq2_xxs, context_tokens: 400000, hot_sessions: 1) == 35225600000 + byte_size_count(b: kv_footprint( + candidate: build_iq2_xxs, context: token_count(count: 400000), hot_sessions: 1)) == 35225600000 } // THE CLAIM THE MODULE EXISTS FOR. At a 400k floor the HIGHER-quality build is REJECTED ON MEMORY @@ -102,13 +110,13 @@ test fn w_kv_at_the_context_floor_is_derived_from_architecture() -> Bool { // says so from the numbers rather than from an argument. test fn w_at_the_400k_floor_the_higher_quality_build_is_rejected_on_memory() -> Bool { match evaluate_candidate(candidate: build_iq3_s, capacity: node_capacity, constraints: floor_400k) { - Rejected { model_name: _, quant_label: _, axis: a } => + ServingCandidateRejected { model_name: _, quant_label: _, axis: a } => match a { - DoesNotFitMemory { required_bytes: _, allocatable_bytes: _ } => true + DoesNotFitMemory { required: _, allocatable: _ } => true ContextBelowFloor { declared: _, floor: _ } => false PrefillBelowFloor { measured: _, floor: _ } => false } - Admissible { candidate: _, resident_bytes: _, headroom_bytes: _ } => false + ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false } } @@ -130,9 +138,9 @@ test fn w_without_the_context_floor_the_higher_quality_build_wins() -> Bool { // the whole exercise removes. test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { let impossible = ServingConstraints { - context_floor_tokens: 8000000, + context_floor: token_count(count: 8000000), hot_sessions: 1, - prefill_floor_tokens_per_second: 0, + prefill_floor: tokens_per_second(count: 0), } match choose_serving_candidate( candidates: installed(), capacity: node_capacity, constraints: impossible) { From 5c64edb5ec934987c357b0406754c703f283deec Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 07:01:06 +0000 Subject: [PATCH 04/22] Delete the fabricated empty-list verdict; type positions as TokenCount and counts as Nat Three findings from review 58087, all upheld. THE FAIL-CLOSED ONE. highest_quality seeded its fold with a helper that MANUFACTURED a verdict when the list was empty: ServingCandidateRejected with model_name "none" and DoesNotFitMemory { 0, 0 }. The review's observation that the only caller can never reach it is precisely what made it dangerous rather than harmless -- it is a synthetic row a later consumer would read as a real model rejected for not fitting a real zero-byte capacity, and nothing in the type marks it invented. A failure arm must refuse, never fabricate. std carries no NonEmptyList, and hand-rolling one here would be a workaround for a missing substrate type rather than a fix. So the fold is made TOTAL by answering with an option instead of a value: highest_quality returns CandidateVerdict? seeded with `none`, and the caller turns Absent into the honest NoCandidateAdmissible. The fabricated constructor is deleted outright, and the now-redundant `length(admissible) == 0` test goes with it -- emptiness had two representations and now has one. PARALLEL REPRESENTATION. DeclaredContext.max_positions and YarnScaling.original_positions were bare Int while choice.dag typed the same concept as TokenCount: one fact, two spellings, free to drift. Both are TokenCount now. factor stays Int, correctly, as a dimensionless ratio. COUNTS. ParameterCounts (all three), NarrowedFrom.rejected_count, CompletenessAnswerable.member_count and ShardedWeights.shard_count admitted negatives; all are Nat. quality_rank stays Int deliberately -- it is an ordinal, not a magnitude. Evidence after the change: selector aggregate true, selector refusal arm true (this is the path the fabricated verdict used to sit on, so it is now executed rather than merely unreachable), population aggregate true, population red arm false. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01HL72ndT9xg2dZEb6J8a2Gr --- dag/gunbc/model/choice.dag | 44 ++++++++++--------- dag/gunbc/model/population.dag | 5 ++- dag/gunbc/model/publication.dag | 14 +++--- ...odel_population_narrowing_witness_test.dag | 8 ++-- 4 files changed, 39 insertions(+), 32 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index cfe8af92b29..17faeb2c13a 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -176,29 +176,31 @@ fn choose_serving_candidate( let verdicts = map(candidates, c => evaluate_candidate(candidate: c, capacity: capacity, constraints: constraints)) let admissible = filter(verdicts, v => verdict_is_admissible(verdict: v)) - match length(admissible) == 0 { - true => NoCandidateAdmissible { rejections: verdicts } - false => ChoseCandidate { verdict: highest_quality(verdicts: admissible) } + match highest_quality(verdicts: admissible) { + Absent => NoCandidateAdmissible { rejections: verdicts } + Present { value: best } => ChoseCandidate { verdict: best } } } -// Maximum by quality rank. Ties keep the earlier candidate, which makes the result stable under -// re-ordering of the input roster rather than dependent on it. -fn highest_quality(verdicts: List) -> CandidateVerdict { - fold(verdicts, head_of(verdicts), (best, v) => - match verdict_quality(verdict: v) > verdict_quality(verdict: best) { - true => v - false => best +// Maximum by quality rank, TOTAL over the empty list because it answers with an option rather than +// a value. The predecessor seeded the fold with a head-or-fabricate helper that manufactured a +// ServingCandidateRejected { model_name: "none", ... DoesNotFitMemory { 0, 0 } } when the list was +// empty. That row was unreachable from the only caller, which is exactly what made it dangerous: it +// is a synthetic verdict a later consumer would read as a real rejection of a real model against a +// real zero-byte capacity, and nothing in the type would mark it as invented. A failure arm must +// refuse rather than fabricate, so emptiness is now carried in the RESULT and the caller turns it +// into the honest NoCandidateAdmissible. The seed is `none`, which needs no candidate to exist. +// +// Ties keep the earlier candidate, so the result is stable under re-ordering of the input roster +// rather than dependent on it. +fn highest_quality(verdicts: List) -> CandidateVerdict? { + fold(verdicts, none, (best, v) => + match best { + Absent => Present { value: v } + Present { value: b } => + match verdict_quality(verdict: v) > verdict_quality(verdict: b) { + true => Present { value: v } + false => Present { value: b } + } }) } - -fn head_of(verdicts: List) -> CandidateVerdict { - match first(verdicts) { - Present { value: v } => v - Absent => ServingCandidateRejected { - model_name: "none" as NonEmptyStr, - quant_label: "none" as NonEmptyStr, - axis: DoesNotFitMemory { required: byte_size(count: 0), allocatable: byte_size(count: 0) }, - } - } -} diff --git a/dag/gunbc/model/population.dag b/dag/gunbc/model/population.dag index 87460d62613..f8453a8328f 100644 --- a/dag/gunbc/model/population.dag +++ b/dag/gunbc/model/population.dag @@ -1,6 +1,7 @@ module gunbc.model.population import std.types { String, Bool, List, NonEmptyStr, Int, Timestamp } +import std.nat { Nat } import gunbc.model.publication { ModelRelease, ReleaseIdentity, DistributionChannel, DistributionObservation, release_identity, release_is_open_weight, distribution_channel_wire, @@ -70,7 +71,7 @@ type ReleaseDiscoverySource { // only population anyone authors directly. type PopulationProvenance = DiscoveredRoot { sources: List } - | NarrowedFrom { parent_stage: PopulationStage, rejected_count: Int } + | NarrowedFrom { parent_stage: PopulationStage, rejected_count: Nat } type ModelPopulation { stage: PopulationStage @@ -127,7 +128,7 @@ fn identity_in(roster: List, candidate: ReleaseIdentity) -> Boo // the axis being asked about. Answering from a single distributor's catalog about the open-weight // universe is the original defect in its general form: a real count over the wrong population. type CompletenessVerdict - = CompletenessAnswerable { member_count: Int } + = CompletenessAnswerable { member_count: Nat } | CompletenessRefused { reason: String } fn open_weight_completeness(population: ModelPopulation) -> CompletenessVerdict { diff --git a/dag/gunbc/model/publication.dag b/dag/gunbc/model/publication.dag index 7f822954831..f0e7202543b 100644 --- a/dag/gunbc/model/publication.dag +++ b/dag/gunbc/model/publication.dag @@ -2,6 +2,8 @@ module gunbc.model.publication import std.types { String, Bool, List, NonEmptyStr, Int, Timestamp } import std.content_hash { ContentHash, serialize_content_hash } +import std.nat { Nat } +import std.measure { TokenCount } // WHAT A MODEL RELEASE IS, modeled independently of anyone who serves, packages or hosts it. // @@ -53,9 +55,9 @@ type WeightLicense // activated is the per-token count for a mixture-of-experts release and equals total for a dense // one. It governs decode bandwidth; total governs whether the release is resident at all. type ParameterCounts { - base_total: Int - activated_per_token: Int - attached_speculative: Int + base_total: Nat + activated_per_token: Nat + attached_speculative: Nat } // WEIGHTS AS A PUBLISHED FACT: a repository, a revision, a license. This is what makes a release @@ -76,10 +78,10 @@ type WeightPublication { // population that requires context is derived from OBSERVED retrieval elsewhere. type ContextScaling = NoScaling - | YarnScaling { original_positions: Int, factor: Int } + | YarnScaling { original_positions: TokenCount, factor: Int } type DeclaredContext { - max_positions: Int + max_positions: TokenCount scaling: ContextScaling } @@ -143,7 +145,7 @@ fn distribution_channel_wire(channel: DistributionChannel) -> String { // nothing about the release and everything about that registry's loader. type ArtifactPackaging = SingleFileWeights { quantization: NonEmptyStr } - | ShardedWeights { quantization: NonEmptyStr, shard_count: Int } + | ShardedWeights { quantization: NonEmptyStr, shard_count: Nat } | PublisherNativeCheckpoint // AN OBSERVATION ABOUT ONE CHANNEL, AND ONLY ONE CHANNEL. diff --git a/dag/test/claim/model/model_population_narrowing_witness_test.dag b/dag/test/claim/model/model_population_narrowing_witness_test.dag index 26cfca470c0..98cbde0c81c 100644 --- a/dag/test/claim/model/model_population_narrowing_witness_test.dag +++ b/dag/test/claim/model/model_population_narrowing_witness_test.dag @@ -1,6 +1,8 @@ module test.claim.model.model_population_narrowing_witness_test import std.types { String, Bool, List, NonEmptyStr, Int, Timestamp } +import std.nat { Nat } +import std.measure { TokenCount, token_count } import gunbc.model.publication { ModelRelease, OpenWeightRelease, ClosedWeightRelease, ReleaseIdentity, WeightPublication, ParameterCounts, DeclaredContext, @@ -45,8 +47,8 @@ data deepseek_v4_flash: ModelRelease = OpenWeightRelease { }, }, declared_context: DeclaredContext { - max_positions: 1048576, - scaling: YarnScaling { original_positions: 65536, factor: 16 }, + max_positions: token_count(count: 1048576), + scaling: YarnScaling { original_positions: token_count(count: 65536), factor: 16 }, }, } @@ -56,7 +58,7 @@ data a_closed_release: ModelRelease = ClosedWeightRelease { family: "Proprietary-XL" as NonEmptyStr, revision: "1" as NonEmptyStr, }, - declared_context: DeclaredContext { max_positions: 200000, scaling: NoScaling }, + declared_context: DeclaredContext { max_positions: token_count(count: 200000), scaling: NoScaling }, } data probed_at: Timestamp = 1788240000 as Timestamp From 2017773c63c61e1cac2ee5841f6dab5feb11187c Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 07:19:11 +0000 Subject: [PATCH 05/22] Regenerate the stage0 std.measure mirror for TokensPerSecond CI cause, not a new feature. Adding TokensPerSecond to std/measure.dag -- which review 58079 correctly required, so that prefill throughput consumes a measure carrier instead of a bare Int -- has a mandatory second half: the seed compiler builds from src/v1/stage0/src/std_measure.rs, a GENERATED mirror of that module. The .dag carried the type and the mirror did not, so the bootstrap build broke. That is why required-witnesses-build failed alongside the floor rather than the floor alone. The hunk is the regen actuator's own output installed verbatim, not hand-written into a file whose header says do-not-edit: claim_executor --required-regen --source-root dag --source-root src/v2 reported `FAIL generated surface drift: std_measure.rs` and named exactly one divergent file. Its candidate tree differs from the checked-in mirror by one hunk of thirteen lines -- the type alias and the two accessors -- at the offset the generator chose. WHY LOCAL VERIFICATION COULD NOT HAVE CAUGHT THIS, recorded because the blind spot is reusable rather than incidental. Every witness in this branch runs against a freshly compiled compiler, never against the mirror, so green-by-execution on a fresh build is structurally silent about seed drift. A change that touches a std module needs the regen check inside the authoring loop; discovering it from a required lane twenty minutes later is the expensive path. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01HL72ndT9xg2dZEb6J8a2Gr --- src/v1/stage0/src/std_measure.rs | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/src/v1/stage0/src/std_measure.rs b/src/v1/stage0/src/std_measure.rs index 4716f07ee13..3a211012575 100644 --- a/src/v1/stage0/src/std_measure.rs +++ b/src/v1/stage0/src/std_measure.rs @@ -936,6 +936,19 @@ pub fn token_count_value(t: TokenCount) -> Nat { measure_count(t.clone()) } +pub type TokensPerSecond = Rc>; + +pub fn tokens_per_second(count: Nat) -> TokensPerSecond { + Rc::new(Measure { + count: count.clone(), + _phantom: std::marker::PhantomData, + }) +} + +pub fn tokens_per_second_count(r: TokensPerSecond) -> Nat { + measure_count(r.clone()) +} + pub fn millicore(count: Nat) -> Millicore { Rc::new(Measure { count: count.clone(), From 58e54230c53986438f4d4d375a0e60f45be18330 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 07:39:24 +0000 Subject: [PATCH 06/22] Unknown is not rejected: the selector refuses rather than answering from what it happens to know Two independent reviews approved the previous shape. Both missed that choose_serving_candidate would answer confidently in two cases where no answer exists, which is the failure this whole module was built to remove. UNKNOWN != REJECTED. A candidate whose KV cost was never measured has not FAILED the memory test; it has not taken it. The old shape had two outcomes -- chose, or nothing admissible -- so an unmeasured candidate had nowhere to land, and a fully measured but worse candidate would win by default with an answer indistinguishable from one where it genuinely won. CandidateVerdict gains ServingCandidateUnanswerable { missing: MissingFact }. For that arm to be REACHABLE rather than decorative the measurements had to become genuinely absent, so kv_per_token and measured_prefill_rate are optional; a zero sentinel would have been the same fabrication deleted in the previous commit. THE ANSWERABILITY RULE. ChoseCandidate is legal only when nothing unresolved could change the optimum, because choosing the best ANSWERABLE candidate is not choosing the best candidate. Any unresolved candidate now yields SelectionUnanswerable { UnresolvedCandidateCouldWin }, and the remedy it names is a measurement rather than a purchase. NO MANUFACTURED CROSS-RELEASE ORDER. Preference is ordinal WITHIN one release, so two admissible releases have no join and a maximum over them does not exist. Returning one anyway invents an ordering the inputs never contained. That case now refuses with CrossReleaseQualityOrderAbsent. The practical consequence is that this selector will NOT choose between DeepSeek V4 Flash and Qwen3.6 today, which is correct: the missing input is operator judgment on real work, not another measurement or more hardware. Nat SUBTRACTION IS TOTAL (review 58094, non-blocking). allocatable and firmware_carveout answered by subtracting unordered operands. Clamping to zero would be the absorbing-fallback shape -- an impossible machine would quietly become a machine with no room and every candidate would be rejected on memory for a reason that was never true -- so both answer with an option and a malformed capacity surfaces as unanswerable, a fact about the INPUTS rather than a verdict about any candidate. distinct_release_count uses the corpus fold/any/append idiom rather than importing guarantee_measurement's distinct_string_count: std carries no distinct, and reaching into a measurement authority for a list primitive from a model-selection module would invert the layering. Evidence: w_all_serving_choice_claims_hold true, w_all_answerability_claims_hold true, w_without_the_context_floor_the_higher_quality_build_wins true (the control separating a quality-maximizing selector from one returning its first argument), w_at_the_400k_floor_the_two_bit_build_is_chosen true. The four new witnesses each CONSTRUCT the state and observe it -- unmeasured, unresolved-blocks-choice, two-releases-refuse, malformed-capacity -- so no Unanswerable arm is a decoration. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01HL72ndT9xg2dZEb6J8a2Gr --- dag/gunbc/model/choice.dag | 202 +++++++++++++++--- .../model/serving_choice_witness_test.dag | 137 +++++++++++- 2 files changed, 294 insertions(+), 45 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index 17faeb2c13a..f3bf7ee9e08 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -40,14 +40,31 @@ type NodeCapacity { runtime_overhead: ByteSize } -fn firmware_carveout(capacity: NodeCapacity) -> ByteSize { - byte_size(count: byte_size_count(b: capacity.nominal) - byte_size_count(b: capacity.kernel_visible)) +// SUBTRACTION OVER Nat TRAPS, so both of these answer with an option rather than assuming their +// operands are ordered. A vendor sheet smaller than the kernel's own reading, or an overhead +// allowance exceeding the machine, is a MALFORMED CAPACITY -- a fact about the inputs, not a verdict +// about any candidate -- and the selector reports it as unanswerable rather than trapping or +// silently clamping to zero. Clamping would be the absorbing-fallback shape: an impossible machine +// would quietly become a machine with no room, and every candidate would be rejected on memory for +// a reason that was never true. +fn firmware_carveout(capacity: NodeCapacity) -> ByteSize? { + match byte_size_count(b: capacity.kernel_visible) > byte_size_count(b: capacity.nominal) { + true => none + false => Present { + value: byte_size(count: byte_size_count(b: capacity.nominal) - byte_size_count(b: capacity.kernel_visible)), + } + } } // The bytes a model may actually occupy: what the kernel can hand out, less what the serving runtime // and OS need to stay alive. Deliberately NOT derived from a live availability sample. -fn allocatable(capacity: NodeCapacity) -> ByteSize { - byte_size(count: byte_size_count(b: capacity.kernel_visible) - byte_size_count(b: capacity.runtime_overhead)) +fn allocatable(capacity: NodeCapacity) -> ByteSize? { + match byte_size_count(b: capacity.runtime_overhead) > byte_size_count(b: capacity.kernel_visible) { + true => none + false => Present { + value: byte_size(count: byte_size_count(b: capacity.kernel_visible) - byte_size_count(b: capacity.runtime_overhead)), + } + } } // ============================ CANDIDATES ============================ @@ -62,20 +79,20 @@ type QuantizedCandidate { weights: ByteSize quality_rank: Int declared_context: TokenCount - kv_per_token: ByteSize - measured_prefill_rate: TokensPerSecond + kv_per_token: ByteSize? + measured_prefill_rate: TokensPerSecond? } // KV scales with context and with how many sessions are simultaneously RESIDENT. Sessions parked to // disk between turns do not count -- their KV is not in memory -- which is why hot_sessions is the // parameter rather than a total session count. -fn kv_footprint(candidate: QuantizedCandidate, context: TokenCount, hot_sessions: Nat) -> ByteSize { - byte_size(count: byte_size_count(b: candidate.kv_per_token) * token_count_value(t: context) * hot_sessions) +fn kv_footprint(per_token: ByteSize, context: TokenCount, hot_sessions: Nat) -> ByteSize { + byte_size(count: byte_size_count(b: per_token) * token_count_value(t: context) * hot_sessions) } -fn resident_footprint(candidate: QuantizedCandidate, context: TokenCount, hot_sessions: Nat) -> ByteSize { - let kv = kv_footprint(candidate: candidate, context: context, hot_sessions: hot_sessions) - byte_size(count: byte_size_count(b: candidate.weights) + byte_size_count(b: kv)) +fn resident_footprint(weights: ByteSize, per_token: ByteSize, context: TokenCount, hot_sessions: Nat) -> ByteSize { + let kv = kv_footprint(per_token: per_token, context: context, hot_sessions: hot_sessions) + byte_size(count: byte_size_count(b: weights) + byte_size_count(b: kv)) } // ============================ CONSTRAINTS ============================ @@ -90,6 +107,23 @@ type ServingConstraints { // ============================ THE DECISION ============================ +// UNKNOWN IS NOT REJECTED, and conflating them is how a selector comes to assert more than it knows. +// A candidate whose KV cost was never measured has not failed the memory test -- it has not TAKEN +// it. Reporting that as a rejection would let a fully-measured but worse candidate win by default, +// and the answer would look identical to one where the better candidate genuinely lost. +type MissingFact + = KvFootprintUnmeasured + | PrefillRateUnmeasured + | CapacityMalformed { detail: String } + +fn missing_fact_wire(missing: MissingFact) -> String { + match missing { + KvFootprintUnmeasured => "kv-footprint-unmeasured" + PrefillRateUnmeasured => "prefill-rate-unmeasured" + CapacityMalformed { detail: d } => join(["capacity-malformed: ", d], "") + } +} + // Why a candidate was rejected, at the grain that tells you what to change. One axis per arm, so a // census over rejections partitions cleanly into "buy hardware", "lower a floor", "wait for a // better artifact". @@ -101,6 +135,7 @@ type RejectionAxis type CandidateVerdict = ServingCandidateAdmissible { candidate: QuantizedCandidate, resident: ByteSize, headroom: ByteSize } | ServingCandidateRejected { model_name: NonEmptyStr, quant_label: NonEmptyStr, axis: RejectionAxis } + | ServingCandidateUnanswerable { model_name: NonEmptyStr, quant_label: NonEmptyStr, missing: MissingFact } // Evaluated in a fixed order so the reported axis is deterministic: capability floors first, then // fit. A candidate that fails several axes reports the first, and the ordering puts the axis the @@ -110,37 +145,62 @@ fn evaluate_candidate( capacity: NodeCapacity, constraints: ServingConstraints, ) -> CandidateVerdict { - let required = resident_footprint( - candidate: candidate, - context: constraints.context_floor, - hot_sessions: constraints.hot_sessions, - ) - let room = allocatable(capacity: capacity) - match token_count_value(t: candidate.declared_context) < token_count_value(t: constraints.context_floor) { - true => ServingCandidateRejected { + match allocatable(capacity: capacity) { + Absent => ServingCandidateUnanswerable { model_name: candidate.model_name, quant_label: candidate.quant_label, - axis: ContextBelowFloor { declared: candidate.declared_context, floor: constraints.context_floor }, + missing: CapacityMalformed { + detail: "declared runtime overhead exceeds the kernel-visible total, so no allocatable figure exists", + }, } - false => match tokens_per_second_count(r: candidate.measured_prefill_rate) < tokens_per_second_count(r: constraints.prefill_floor) { - true => ServingCandidateRejected { + Present { value: room } => match candidate.kv_per_token { + Absent => ServingCandidateUnanswerable { model_name: candidate.model_name, quant_label: candidate.quant_label, - axis: PrefillBelowFloor { - measured: candidate.measured_prefill_rate, - floor: constraints.prefill_floor, - }, + missing: KvFootprintUnmeasured, } - false => match byte_size_count(b: required) > byte_size_count(b: room) { + Present { value: per_token } => match candidate.measured_prefill_rate { + Absent => ServingCandidateUnanswerable { + model_name: candidate.model_name, + quant_label: candidate.quant_label, + missing: PrefillRateUnmeasured, + } + Present { value: prefill } => { + let required = resident_footprint( + weights: candidate.weights, + per_token: per_token, + context: constraints.context_floor, + hot_sessions: constraints.hot_sessions, + ) + match token_count_value(t: candidate.declared_context) < token_count_value(t: constraints.context_floor) { true => ServingCandidateRejected { model_name: candidate.model_name, quant_label: candidate.quant_label, - axis: DoesNotFitMemory { required: required, allocatable: room }, + axis: ContextBelowFloor { declared: candidate.declared_context, floor: constraints.context_floor }, } - false => ServingCandidateAdmissible { - candidate: candidate, - resident: required, - headroom: byte_size(count: byte_size_count(b: room) - byte_size_count(b: required)), + false => match tokens_per_second_count(r: prefill) < tokens_per_second_count(r: constraints.prefill_floor) { + true => ServingCandidateRejected { + model_name: candidate.model_name, + quant_label: candidate.quant_label, + axis: PrefillBelowFloor { + measured: prefill, + floor: constraints.prefill_floor, + }, + } + false => match byte_size_count(b: required) > byte_size_count(b: room) { + true => ServingCandidateRejected { + model_name: candidate.model_name, + quant_label: candidate.quant_label, + axis: DoesNotFitMemory { required: required, allocatable: room }, + } + false => ServingCandidateAdmissible { + candidate: candidate, + resident: required, + headroom: byte_size(count: byte_size_count(b: room) - byte_size_count(b: required)), + } + } + } + } } } } @@ -151,6 +211,7 @@ fn verdict_is_admissible(verdict: CandidateVerdict) -> Bool { match verdict { ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => true ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false + ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: _ } => false } } @@ -158,6 +219,7 @@ fn verdict_quality(verdict: CandidateVerdict) -> Int { match verdict { ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => c.quality_rank ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => 0 - 1 + ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: _ } => 0 - 1 } } @@ -167,6 +229,32 @@ fn verdict_quality(verdict: CandidateVerdict) -> Int { type ServingChoice = ChoseCandidate { verdict: CandidateVerdict } | NoCandidateAdmissible { rejections: List } + | SelectionUnanswerable { cause: UnanswerableCause, evaluations: List } + +// WHY NO ANSWER EXISTS YET, as distinct from no candidate qualifying. Both refuse; they tell the +// operator to do completely different things. +// +// UnresolvedCandidateCouldWin is the answerability rule. Choosing the best ANSWERABLE candidate is +// not choosing the best candidate: if an unmeasured entry might beat the apparent winner once +// measured, the honest output is that the question is open, and the remedy is a measurement rather +// than a purchase. +// +// CrossReleaseQualityOrderAbsent is the consequence of preference being ordinal WITHIN a release. +// Two releases each carrying their own ordinal have no join, so a maximum over them does not exist +// and returning one MANUFACTURES an ordering the inputs never contained. The remedy is an operator +// policy or accumulated head-to-head evidence, not a tie-break rule invented here. +type UnanswerableCause + = UnresolvedCandidateCouldWin { unresolved_count: Nat } + | CrossReleaseQualityOrderAbsent { distinct_release_count: Nat } + +fn unanswerable_cause_wire(cause: UnanswerableCause) -> String { + match cause { + UnresolvedCandidateCouldWin { unresolved_count: n } => + join(["unresolved candidates could change the optimum: ", to_string(n)], "") + CrossReleaseQualityOrderAbsent { distinct_release_count: n } => + join(["no cross-release quality order exists over ", to_string(n), " distinct releases"], "") + } +} fn choose_serving_candidate( candidates: List, @@ -175,10 +263,54 @@ fn choose_serving_candidate( ) -> ServingChoice { let verdicts = map(candidates, c => evaluate_candidate(candidate: c, capacity: capacity, constraints: constraints)) + let unresolved = filter(verdicts, v => verdict_is_unanswerable(verdict: v)) let admissible = filter(verdicts, v => verdict_is_admissible(verdict: v)) - match highest_quality(verdicts: admissible) { - Absent => NoCandidateAdmissible { rejections: verdicts } - Present { value: best } => ChoseCandidate { verdict: best } + match length(unresolved) > 0 { + true => SelectionUnanswerable { + cause: UnresolvedCandidateCouldWin { unresolved_count: length(unresolved) }, + evaluations: verdicts, + } + false => match distinct_release_count(verdicts: admissible) > 1 { + true => SelectionUnanswerable { + cause: CrossReleaseQualityOrderAbsent { distinct_release_count: distinct_release_count(verdicts: admissible) }, + evaluations: verdicts, + } + false => match highest_quality(verdicts: admissible) { + Absent => NoCandidateAdmissible { rejections: verdicts } + Present { value: best } => ChoseCandidate { verdict: best } + } + } + } +} + +fn verdict_is_unanswerable(verdict: CandidateVerdict) -> Bool { + match verdict { + ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: _ } => true + ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false + ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false + } +} + +// How many DISTINCT releases the admissible set spans. Preference is ordinal within one release, so +// more than one release means the maximum this function would return does not exist. +// +// The dedupe is the corpus's fold/any/append idiom rather than a call to guarantee_measurement's +// distinct_string_count: std carries no distinct, and reaching into the guarantee-measurement module +// for a list primitive would invert the layering -- a model-selection authority would then depend on +// a measurement authority for something neither owns. +fn distinct_release_count(verdicts: List) -> Nat { + length(fold(map(verdicts, v => verdict_model_name(verdict: v)), [], (acc, n) => + match any(acc, e => e == n) { + true => acc + false => append(acc, [n]) + })) +} + +fn verdict_model_name(verdict: CandidateVerdict) -> String { + match verdict { + ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => c.model_name as String + ServingCandidateRejected { model_name: m, quant_label: _, axis: _ } => m as String + ServingCandidateUnanswerable { model_name: m, quant_label: _, missing: _ } => m as String } } diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index 4a06f1c65c5..a4c303cf704 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -8,10 +8,13 @@ import std.measure { TokensPerSecond, tokens_per_second, } import gunbc.model.choice { + PrefillRateUnmeasured, NodeCapacity, QuantizedCandidate, ServingConstraints, CandidateVerdict, ServingCandidateAdmissible, ServingCandidateRejected, RejectionAxis, DoesNotFitMemory, ContextBelowFloor, PrefillBelowFloor, - ServingChoice, ChoseCandidate, NoCandidateAdmissible, + ServingChoice, ChoseCandidate, NoCandidateAdmissible, SelectionUnanswerable, + UnanswerableCause, UnresolvedCandidateCouldWin, CrossReleaseQualityOrderAbsent, + ServingCandidateUnanswerable, MissingFact, KvFootprintUnmeasured, CapacityMalformed, evaluate_candidate, choose_serving_candidate, allocatable, firmware_carveout, kv_footprint, } @@ -44,8 +47,8 @@ data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { weights: byte_size(count: 86720111200), quality_rank: 2, declared_context: token_count(count: 1048576), - kv_per_token: kv_per_token, - measured_prefill_rate: tokens_per_second(count: 197), + kv_per_token: Present { value: kv_per_token }, + measured_prefill_rate: Present { value: tokens_per_second(count: 197) }, } data build_iq3_s: QuantizedCandidate = QuantizedCandidate { @@ -54,8 +57,8 @@ data build_iq3_s: QuantizedCandidate = QuantizedCandidate { weights: byte_size(count: 116100000000), quality_rank: 3, declared_context: token_count(count: 1048576), - kv_per_token: kv_per_token, - measured_prefill_rate: tokens_per_second(count: 197), + kv_per_token: Present { value: kv_per_token }, + measured_prefill_rate: Present { value: tokens_per_second(count: 197) }, } fn installed() -> List { @@ -78,11 +81,13 @@ data floor_8k: ServingConstraints = ServingConstraints { fn chosen_label(choice: ServingChoice) -> String { match choice { + SelectionUnanswerable { cause: _, evaluations: _ } => "unanswerable" NoCandidateAdmissible { rejections: _ } => "none" ChoseCandidate { verdict: v } => match v { ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => c.quant_label as String ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => "none" + ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: _ } => "none" } } } @@ -92,16 +97,23 @@ fn chosen_label(choice: ServingChoice) -> String { // The three capacity grains are DIFFERENT, and only one of them is a capacity. Quoting the vendor // figure is what made a 128.1 GB build look feasible. test fn w_capacity_grains_differ_and_only_kernel_visible_is_spendable() -> Bool { - byte_size_count(b: firmware_carveout(capacity: node_capacity)) == 6778802176 - && byte_size_count(b: allocatable(capacity: node_capacity)) == 127660151296 - && byte_size_count(b: allocatable(capacity: node_capacity)) - < byte_size_count(b: node_capacity.nominal) + match firmware_carveout(capacity: node_capacity) { + Absent => false + Present { value: c } => + match allocatable(capacity: node_capacity) { + Absent => false + Present { value: a } => + byte_size_count(b: c) == 6778802176 + && byte_size_count(b: a) == 127660151296 + && byte_size_count(b: a) < byte_size_count(b: node_capacity.nominal) + } + } } // KV at the operator's floor, from the model's own dimensions: 400,000 * 88,064 = 35.2 GB. test fn w_kv_at_the_context_floor_is_derived_from_architecture() -> Bool { byte_size_count(b: kv_footprint( - candidate: build_iq2_xxs, context: token_count(count: 400000), hot_sessions: 1)) == 35225600000 + per_token: kv_per_token, context: token_count(count: 400000), hot_sessions: 1)) == 35225600000 } // THE CLAIM THE MODULE EXISTS FOR. At a 400k floor the HIGHER-quality build is REJECTED ON MEMORY @@ -117,6 +129,7 @@ test fn w_at_the_400k_floor_the_higher_quality_build_is_rejected_on_memory() -> PrefillBelowFloor { measured: _, floor: _ } => false } ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false + ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: _ } => false } } @@ -146,6 +159,7 @@ test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { candidates: installed(), capacity: node_capacity, constraints: impossible) { NoCandidateAdmissible { rejections: r } => length(r) == 2 ChoseCandidate { verdict: _ } => false + SelectionUnanswerable { cause: _, evaluations: _ } => false } } @@ -157,3 +171,106 @@ test fn w_all_serving_choice_claims_hold() -> Bool { && w_without_the_context_floor_the_higher_quality_build_wins() && w_nothing_admissible_refuses_and_reports_every_rejection() } + +// ======================= THE ANSWERABILITY CLAIMS ======================= +// Each arm below must be REACHABLE. An Unanswerable variant nothing can construct would be a +// decoration that reads as coverage, so every one of these builds the state and observes it. + +// A candidate whose KV cost was never measured has not FAILED the memory test -- it has not taken +// it. The selector must refuse rather than hand the win to the measured candidate by default. +data build_unmeasured: QuantizedCandidate = QuantizedCandidate { + model_name: "DeepSeek-V4-Flash-0731" as NonEmptyStr, + quant_label: "UD-IQ4_XS" as NonEmptyStr, + weights: byte_size(count: 136700000000), + quality_rank: 4, + declared_context: token_count(count: 1048576), + kv_per_token: none, + measured_prefill_rate: Present { value: tokens_per_second(count: 197) }, +} + +test fn w_an_unmeasured_candidate_is_unanswerable_not_rejected() -> Bool { + match evaluate_candidate(candidate: build_unmeasured, capacity: node_capacity, constraints: floor_400k) { + ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: m } => + match m { + KvFootprintUnmeasured => true + PrefillRateUnmeasured => false + CapacityMalformed { detail: _ } => false + } + ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false + ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false + } +} + +// THE LOAD-BEARING CLAIM. A measured, admissible candidate must NOT win while an unmeasured one is +// outstanding: choosing the best ANSWERABLE candidate is not choosing the best candidate. +test fn w_selection_refuses_while_an_unresolved_candidate_could_win() -> Bool { + match choose_serving_candidate( + candidates: [build_iq2_xxs, build_unmeasured], capacity: node_capacity, constraints: floor_400k) { + SelectionUnanswerable { cause: c, evaluations: _ } => + match c { + UnresolvedCandidateCouldWin { unresolved_count: n } => n == 1 + CrossReleaseQualityOrderAbsent { distinct_release_count: _ } => false + } + ChoseCandidate { verdict: _ } => false + NoCandidateAdmissible { rejections: _ } => false + } +} + +// Preference is ordinal WITHIN a release, so two admissible releases have no join and a maximum over +// them does not exist. Returning one would manufacture an ordering the inputs never contained. +data other_release: QuantizedCandidate = QuantizedCandidate { + model_name: "Qwen3.6-35B" as NonEmptyStr, + quant_label: "Q8_0" as NonEmptyStr, + weights: byte_size(count: 22600000000), + quality_rank: 3, + declared_context: token_count(count: 1048576), + kv_per_token: Present { value: byte_size(count: 20000) }, + measured_prefill_rate: Present { value: tokens_per_second(count: 900) }, +} + +test fn w_two_admissible_releases_refuse_for_want_of_a_cross_release_order() -> Bool { + match choose_serving_candidate( + candidates: [build_iq2_xxs, other_release], capacity: node_capacity, constraints: floor_400k) { + SelectionUnanswerable { cause: c, evaluations: _ } => + match c { + CrossReleaseQualityOrderAbsent { distinct_release_count: n } => n == 2 + UnresolvedCandidateCouldWin { unresolved_count: _ } => false + } + ChoseCandidate { verdict: _ } => false + NoCandidateAdmissible { rejections: _ } => false + } +} + +// A malformed capacity is a fact about the INPUTS, not a verdict about a candidate. Clamping the +// Nat subtraction to zero would turn an impossible machine into a machine with no room and reject +// every candidate on memory for a reason that was never true. +data malformed_capacity: NodeCapacity = NodeCapacity { + node_name: "impossible" as NonEmptyStr, + nominal: byte_size(count: 137438953472), + kernel_visible: byte_size(count: 130660151296), + runtime_overhead: byte_size(count: 200000000000), +} + +test fn w_malformed_capacity_is_unanswerable_not_a_memory_rejection() -> Bool { + match allocatable(capacity: malformed_capacity) { + Present { value: _ } => false + Absent => + match evaluate_candidate(candidate: build_iq2_xxs, capacity: malformed_capacity, constraints: floor_400k) { + ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: m } => + match m { + CapacityMalformed { detail: _ } => true + KvFootprintUnmeasured => false + PrefillRateUnmeasured => false + } + ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false + ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false + } + } +} + +test fn w_all_answerability_claims_hold() -> Bool { + w_an_unmeasured_candidate_is_unanswerable_not_rejected() + && w_selection_refuses_while_an_unresolved_candidate_could_win() + && w_two_admissible_releases_refuse_for_want_of_a_cross_release_order() + && w_malformed_capacity_is_unanswerable_not_a_memory_rejection() +} From 3c2e347fd8cb52f048075ab3819928f25cc158f2 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 08:24:39 +0000 Subject: [PATCH 07/22] A must-be-false probe has no floor arm, and the roster it would need forbids it The floor refused at 58e5423 on one row: required-floor: FAIL ...w_red_arm_forbidden_edge_must_evaluate_false returned Bool(false) verdict=FloorRefused unexpected_failures=1 The tree compiled; the row was authored to return false on purpose. The floor's only arm for that is v2.workflow.floor_expected_red, and that roster's own header rules it out -- "a row belongs here only while someone is fixing it", "an identity sitting here indefinitely is a defect nobody owns wearing a receipt". A permanent discriminating control is not debt, so enrolling it would have made the roster the skip list it is carefully not. Restated in the corpus idiom instead -- a positive assertion over a red input, the shape of ..._wrong_fixture_refuses_holds. w_population_membership_discriminates_in_ both_directions asserts a release the root population does not carry is not held, and the one it does carry is. That keeps what the old row was after (population_holds is not constantly true, so the claims above it are not vacuous) while reaching a terminal verdict. The forbidden edge itself was never carried by that row: RED 1 asserts it directly. Also removes three byte-identical declarations appended at the tail of choice.dag during the tie work -- verdict_is_unanswerable, distinct_release_count, verdict_model_name -- which the resolver refused as "a second declaration of one name silently replaced the first", and completes the MissingFact rename in the two serving-choice match sites that still listed the pre-split arm. Verified by execution, all seven returning true against the deduped tree: w_all_population_claims_hold, w_population_membership_discriminates_in_both_directions, w_all_serving_choice_claims_hold, w_all_answerability_claims_hold, w_tied_quality_ranks_refuse_in_both_roster_orders, w_a_declared_context_never_qualifies_without_a_retrieval_receipt, w_the_higher_quality_candidate_wins_when_both_are_admissible Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01HL72ndT9xg2dZEb6J8a2Gr --- dag/gunbc/model/choice.dag | 176 +++++++++++++----- ...odel_population_narrowing_witness_test.dag | 20 +- .../model/serving_choice_witness_test.dag | 167 +++++++++++++++-- 3 files changed, 296 insertions(+), 67 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index f3bf7ee9e08..05340778ea0 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -79,8 +79,34 @@ type QuantizedCandidate { weights: ByteSize quality_rank: Int declared_context: TokenCount + semantic_context_verified_to: TokenCount? + prefill_observations: List kv_per_token: ByteSize? - measured_prefill_rate: TokensPerSecond? +} + +// PREFILL RATE IS NOT A SCALAR, and this repository's own measurements falsified the scalar that +// used to sit here. The same realization measured 253 tok/s at 160,060 tokens, 197 at 255,061 and +// 135 at 400,060: attention cost grows with depth, so a single rate silently describes whichever +// depth the measurer happened to use. A consumer asking "is this fast enough at my floor" needs an +// observation AT OR BEYOND that floor; anything shallower flatters the candidate. +type PrefillObservation { + depth: TokenCount + rate: TokensPerSecond +} + +// The slowest rate observed at or beyond the floor, or Absent when nothing reaches that depth. +// Slowest rather than fastest because a floor is a promise about the worst case a caller will meet. +fn prefill_rate_at_floor(observations: List, floor: TokenCount) -> TokensPerSecond? { + let deep = filter(observations, o => token_count_value(t: o.depth) >= token_count_value(t: floor)) + fold(deep, none, (worst, o) => + match worst { + Absent => Present { value: o.rate } + Present { value: w } => + match tokens_per_second_count(r: o.rate) < tokens_per_second_count(r: w) { + true => Present { value: o.rate } + false => Present { value: w } + } + }) } // KV scales with context and with how many sessions are simultaneously RESIDENT. Sessions parked to @@ -113,13 +139,21 @@ type ServingConstraints { // and the answer would look identical to one where the better candidate genuinely lost. type MissingFact = KvFootprintUnmeasured - | PrefillRateUnmeasured + | PrefillRateUnmeasuredAtFloor { floor: TokenCount } + | SemanticContextUnverifiedAtFloor { floor: TokenCount, declared: TokenCount } | CapacityMalformed { detail: String } fn missing_fact_wire(missing: MissingFact) -> String { match missing { KvFootprintUnmeasured => "kv-footprint-unmeasured" - PrefillRateUnmeasured => "prefill-rate-unmeasured" + PrefillRateUnmeasuredAtFloor { floor: f } => + join(["no prefill observation at or beyond ", to_string(token_count_value(t: f))], "") + SemanticContextUnverifiedAtFloor { floor: f, declared: d } => + join([ + "retrieval unverified at ", to_string(token_count_value(t: f)), + "; the publisher declares ", to_string(token_count_value(t: d)), + " but a declaration is a claim, not evidence", + ], "") CapacityMalformed { detail: d } => join(["capacity-malformed: ", d], "") } } @@ -140,6 +174,17 @@ type CandidateVerdict // Evaluated in a fixed order so the reported axis is deterministic: capability floors first, then // fit. A candidate that fails several axes reports the first, and the ordering puts the axis the // operator can act on soonest at the front. +// EVALUATION ORDER IS LOAD-BEARING, not stylistic. Memory fit is decidable from weights, KV cost and +// capacity alone, so it is settled FIRST: a candidate that cannot be resident does not need a +// retrieval receipt or a prefill observation, and demanding evidence before rejecting it would turn +// a decidable no into an unanswerable maybe. Only once a candidate could actually serve do we ask +// whether it is EVIDENCED to serve at this floor. +// +// The context test consumes semantic_context_verified_to -- an observed retrieval depth -- and NEVER +// declared_context. A publisher declaring 1,048,576 positions has made a claim about what the +// architecture admits, not a measurement of what the model retrieves, and the gap between them is +// exactly where silent degradation lives. Admitting on the declaration would reintroduce the rung +// error the publication layer exists to prevent, one module downstream of where it was removed. fn evaluate_candidate( candidate: QuantizedCandidate, capacity: NodeCapacity, @@ -159,50 +204,62 @@ fn evaluate_candidate( quant_label: candidate.quant_label, missing: KvFootprintUnmeasured, } - Present { value: per_token } => match candidate.measured_prefill_rate { - Absent => ServingCandidateUnanswerable { - model_name: candidate.model_name, - quant_label: candidate.quant_label, - missing: PrefillRateUnmeasured, - } - Present { value: prefill } => { - let required = resident_footprint( - weights: candidate.weights, - per_token: per_token, - context: constraints.context_floor, - hot_sessions: constraints.hot_sessions, - ) - match token_count_value(t: candidate.declared_context) < token_count_value(t: constraints.context_floor) { - true => ServingCandidateRejected { - model_name: candidate.model_name, - quant_label: candidate.quant_label, - axis: ContextBelowFloor { declared: candidate.declared_context, floor: constraints.context_floor }, - } - false => match tokens_per_second_count(r: prefill) < tokens_per_second_count(r: constraints.prefill_floor) { + Present { value: per_token } => { + let required = resident_footprint( + weights: candidate.weights, + per_token: per_token, + context: constraints.context_floor, + hot_sessions: constraints.hot_sessions, + ) + match byte_size_count(b: required) > byte_size_count(b: room) { true => ServingCandidateRejected { model_name: candidate.model_name, quant_label: candidate.quant_label, - axis: PrefillBelowFloor { - measured: prefill, - floor: constraints.prefill_floor, - }, + axis: DoesNotFitMemory { required: required, allocatable: room }, } - false => match byte_size_count(b: required) > byte_size_count(b: room) { - true => ServingCandidateRejected { + false => match candidate.semantic_context_verified_to { + Absent => ServingCandidateUnanswerable { model_name: candidate.model_name, quant_label: candidate.quant_label, - axis: DoesNotFitMemory { required: required, allocatable: room }, - } - false => ServingCandidateAdmissible { - candidate: candidate, - resident: required, - headroom: byte_size(count: byte_size_count(b: room) - byte_size_count(b: required)), + missing: SemanticContextUnverifiedAtFloor { + floor: constraints.context_floor, + declared: candidate.declared_context, + }, } + Present { value: verified } => + match token_count_value(t: verified) < token_count_value(t: constraints.context_floor) { + true => ServingCandidateRejected { + model_name: candidate.model_name, + quant_label: candidate.quant_label, + axis: ContextBelowFloor { declared: verified, floor: constraints.context_floor }, + } + false => match prefill_rate_at_floor( + observations: candidate.prefill_observations, + floor: constraints.context_floor, + ) { + Absent => ServingCandidateUnanswerable { + model_name: candidate.model_name, + quant_label: candidate.quant_label, + missing: PrefillRateUnmeasuredAtFloor { floor: constraints.context_floor }, + } + Present { value: prefill } => + match tokens_per_second_count(r: prefill) < tokens_per_second_count(r: constraints.prefill_floor) { + true => ServingCandidateRejected { + model_name: candidate.model_name, + quant_label: candidate.quant_label, + axis: PrefillBelowFloor { measured: prefill, floor: constraints.prefill_floor }, + } + false => ServingCandidateAdmissible { + candidate: candidate, + resident: required, + headroom: byte_size(count: byte_size_count(b: room) - byte_size_count(b: required)), + } + } + } + } } } } - } - } } } } @@ -246,6 +303,7 @@ type ServingChoice type UnanswerableCause = UnresolvedCandidateCouldWin { unresolved_count: Nat } | CrossReleaseQualityOrderAbsent { distinct_release_count: Nat } + | QualityRankTie { tied_count: Nat, rank: Int } fn unanswerable_cause_wire(cause: UnanswerableCause) -> String { match cause { @@ -253,6 +311,8 @@ fn unanswerable_cause_wire(cause: UnanswerableCause) -> String { join(["unresolved candidates could change the optimum: ", to_string(n)], "") CrossReleaseQualityOrderAbsent { distinct_release_count: n } => join(["no cross-release quality order exists over ", to_string(n), " distinct releases"], "") + QualityRankTie { tied_count: n, rank: r } => + join([to_string(n), " admissible candidates share rank ", to_string(r), ", so no maximum exists"], "") } } @@ -275,9 +335,21 @@ fn choose_serving_candidate( cause: CrossReleaseQualityOrderAbsent { distinct_release_count: distinct_release_count(verdicts: admissible) }, evaluations: verdicts, } - false => match highest_quality(verdicts: admissible) { + false => match top_quality_rank(verdicts: admissible) { Absent => NoCandidateAdmissible { rejections: verdicts } - Present { value: best } => ChoseCandidate { verdict: best } + Present { value: rank } => { + let tied = verdicts_at_rank(verdicts: admissible, rank: rank) + match length(tied) > 1 { + true => SelectionUnanswerable { + cause: QualityRankTie { tied_count: length(tied), rank: rank }, + evaluations: verdicts, + } + false => match first(tied) { + Absent => NoCandidateAdmissible { rejections: verdicts } + Present { value: best } => ChoseCandidate { verdict: best } + } + } + } } } } @@ -320,19 +392,29 @@ fn verdict_model_name(verdict: CandidateVerdict) -> String { // empty. That row was unreachable from the only caller, which is exactly what made it dangerous: it // is a synthetic verdict a later consumer would read as a real rejection of a real model against a // real zero-byte capacity, and nothing in the type would mark it as invented. A failure arm must -// refuse rather than fabricate, so emptiness is now carried in the RESULT and the caller turns it -// into the honest NoCandidateAdmissible. The seed is `none`, which needs no candidate to exist. +// refuse rather than fabricate, so emptiness is carried in the RESULT and the caller turns it into +// the honest NoCandidateAdmissible. // -// Ties keep the earlier candidate, so the result is stable under re-ordering of the input roster -// rather than dependent on it. -fn highest_quality(verdicts: List) -> CandidateVerdict? { +// TIES ARE NOT BROKEN HERE. An earlier revision kept the earlier candidate and its annotation +// claimed that made the result stable under re-ordering -- which is exactly backwards, since keeping +// the earlier candidate is the one rule whose output DOES change when the roster is reversed. Two +// candidates at equal rank have no maximum, and inventing one is the same fabrication this module +// refuses across releases, applied within a release. So this returns the top rank and how many share +// it, and the caller refuses when more than one does. +fn top_quality_rank(verdicts: List) -> Int? { fold(verdicts, none, (best, v) => match best { - Absent => Present { value: v } + Absent => Present { value: verdict_quality(verdict: v) } Present { value: b } => - match verdict_quality(verdict: v) > verdict_quality(verdict: b) { - true => Present { value: v } + match verdict_quality(verdict: v) > b { + true => Present { value: verdict_quality(verdict: v) } false => Present { value: b } } }) } + +fn verdicts_at_rank(verdicts: List, rank: Int) -> List { + filter(verdicts, v => verdict_quality(verdict: v) == rank) +} + + diff --git a/dag/test/claim/model/model_population_narrowing_witness_test.dag b/dag/test/claim/model/model_population_narrowing_witness_test.dag index 98cbde0c81c..e45a14b0b9e 100644 --- a/dag/test/claim/model/model_population_narrowing_witness_test.dag +++ b/dag/test/claim/model/model_population_narrowing_witness_test.dag @@ -196,10 +196,18 @@ test fn w_all_population_claims_hold() -> Bool { && w_a_narrowed_stage_refuses_completeness_about_the_open_weight_universe() } -// THE RED ARM, kept enrolled. It asserts the FORBIDDEN edge -- that a channel-absent observation -// removes the release from the open-weight population -- and must therefore evaluate FALSE. Without -// it every claim above is a conjunction of things that happen to be true, and nothing demonstrates -// the carrier can fail. If this ever returns true, the structural separation has been undone. -test fn w_red_arm_forbidden_edge_must_evaluate_false() -> Bool { - !population_holds(population: root(), id: deepseek_v4_flash_id) +// THE DISCRIMINATING CONTROL, kept enrolled. `population_holds` must not be constantly true, or +// every claim above is a conjunction of things that hold by vacuity rather than by the structural +// separation. It asserts both directions in one row: a release the root population does not carry +// is NOT held, and the one it does carry IS. The forbidden edge -- a channel-absent observation +// removing a release -- is asserted directly by RED 1 above; what was missing, and is supplied +// here, is evidence that the membership predicate can answer false at all. +test fn w_population_membership_discriminates_in_both_directions() -> Bool { + let stranger = ReleaseIdentity { + publisher: "nobody" as NonEmptyStr, + family: "Never-Discovered" as NonEmptyStr, + revision: "1" as NonEmptyStr, + } + !population_holds(population: root(), id: stranger) + && population_holds(population: root(), id: deepseek_v4_flash_id) } diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index a4c303cf704..4da414a8504 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -8,12 +8,13 @@ import std.measure { TokensPerSecond, tokens_per_second, } import gunbc.model.choice { - PrefillRateUnmeasured, + PrefillRateUnmeasuredAtFloor, SemanticContextUnverifiedAtFloor, + PrefillObservation, prefill_rate_at_floor, NodeCapacity, QuantizedCandidate, ServingConstraints, CandidateVerdict, ServingCandidateAdmissible, ServingCandidateRejected, RejectionAxis, DoesNotFitMemory, ContextBelowFloor, PrefillBelowFloor, ServingChoice, ChoseCandidate, NoCandidateAdmissible, SelectionUnanswerable, - UnanswerableCause, UnresolvedCandidateCouldWin, CrossReleaseQualityOrderAbsent, + UnanswerableCause, UnresolvedCandidateCouldWin, CrossReleaseQualityOrderAbsent, QualityRankTie, ServingCandidateUnanswerable, MissingFact, KvFootprintUnmeasured, CapacityMalformed, evaluate_candidate, choose_serving_candidate, allocatable, firmware_carveout, kv_footprint, @@ -47,8 +48,13 @@ data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { weights: byte_size(count: 86720111200), quality_rank: 2, declared_context: token_count(count: 1048576), + semantic_context_verified_to: Present { value: token_count(count: 400060) }, + prefill_observations: [ + PrefillObservation { depth: token_count(count: 160060), rate: tokens_per_second(count: 253) }, + PrefillObservation { depth: token_count(count: 255061), rate: tokens_per_second(count: 197) }, + PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, + ], kv_per_token: Present { value: kv_per_token }, - measured_prefill_rate: Present { value: tokens_per_second(count: 197) }, } data build_iq3_s: QuantizedCandidate = QuantizedCandidate { @@ -57,8 +63,9 @@ data build_iq3_s: QuantizedCandidate = QuantizedCandidate { weights: byte_size(count: 116100000000), quality_rank: 3, declared_context: token_count(count: 1048576), + semantic_context_verified_to: none, + prefill_observations: [], kv_per_token: Present { value: kv_per_token }, - measured_prefill_rate: Present { value: tokens_per_second(count: 197) }, } fn installed() -> List { @@ -138,12 +145,81 @@ test fn w_at_the_400k_floor_the_two_bit_build_is_chosen() -> Bool { candidates: installed(), capacity: node_capacity, constraints: floor_400k)) == "IQ2_XXS" } -// THE DISCRIMINATING CONTROL. Drop the context floor and the selector must flip to the HIGHER -// quality build -- proving it maximizes quality and is not simply always picking the smaller file. -// Without this, every claim above is satisfied by a function that returns the first candidate. -test fn w_without_the_context_floor_the_higher_quality_build_wins() -> Bool { +// THE DISCRIMINATING CONTROL, on a CONSTRUCTED pair rather than the installed builds. +// +// It cannot use the real ones any more, and the reason is itself the point: the installed IQ3_S has +// no retrieval receipt at any depth, so at a low floor it is UNANSWERABLE rather than a winner, and +// the selector refuses instead of ranking. That is the honest state -- what was probed on the fleet +// was whether the runtime ACCEPTS a num_ctx value, which a two-token request answers without ever +// allocating the cache, so it established nothing about retrieval. +// +// These two are declared fixtures exercising the ordering rule, NOT fleet observations. They share a +// release, so a maximum over them exists, and they differ only in quality_rank and weights. Without +// this control every other claim here is satisfied by a function that returns its first argument. +data fixture_low_rank: QuantizedCandidate = QuantizedCandidate { + model_name: "Fixture-Release" as NonEmptyStr, + quant_label: "LOW" as NonEmptyStr, + weights: byte_size(count: 20000000000), + quality_rank: 1, + declared_context: token_count(count: 1048576), + semantic_context_verified_to: Present { value: token_count(count: 400060) }, + prefill_observations: [ + PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + ], + kv_per_token: Present { value: byte_size(count: 20000) }, +} + +data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { + model_name: "Fixture-Release" as NonEmptyStr, + quant_label: "HIGH" as NonEmptyStr, + weights: byte_size(count: 30000000000), + quality_rank: 9, + declared_context: token_count(count: 1048576), + semantic_context_verified_to: Present { value: token_count(count: 400060) }, + prefill_observations: [ + PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 400) }, + ], + kv_per_token: Present { value: byte_size(count: 20000) }, +} + +test fn w_the_higher_quality_candidate_wins_when_both_are_admissible() -> Bool { chosen_label(choice: choose_serving_candidate( - candidates: installed(), capacity: node_capacity, constraints: floor_8k)) == "UD-IQ3_S" + candidates: [fixture_low_rank, fixture_high_rank], + capacity: node_capacity, constraints: floor_8k)) == "HIGH" + && chosen_label(choice: choose_serving_candidate( + candidates: [fixture_high_rank, fixture_low_rank], + capacity: node_capacity, constraints: floor_8k)) == "HIGH" +} + +// A DECLARATION IS NOT EVIDENCE. The installed IQ3_S declares 1,048,576 positions and has no +// retrieval receipt; at a floor it FITS, it must come back unanswerable rather than admissible. +// This is the rung error the publication layer removed, re-tested one module downstream where the +// selector could have quietly reintroduced it. +test fn w_a_declared_context_never_qualifies_without_a_retrieval_receipt() -> Bool { + match evaluate_candidate(candidate: build_iq3_s, capacity: node_capacity, constraints: floor_8k) { + ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: m } => + match m { + SemanticContextUnverifiedAtFloor { floor: _, declared: d } => + token_count_value(t: d) == 1048576 + KvFootprintUnmeasured => false + PrefillRateUnmeasuredAtFloor { floor: _ } => false + CapacityMalformed { detail: _ } => false + } + ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false + ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false + } +} + +// A prefill observation SHALLOWER than the floor does not qualify the floor: rate falls with depth +// (253 -> 197 -> 135 measured on one realization), so a shallow reading flatters the candidate. +test fn w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() -> Bool { + match prefill_rate_at_floor( + observations: [PrefillObservation { depth: token_count(count: 160060), rate: tokens_per_second(count: 253) }], + floor: token_count(count: 400000), + ) { + Absent => true + Present { value: _ } => false + } } // When nothing is admissible the selector REFUSES and hands back every rejection, so the binding @@ -168,7 +244,10 @@ test fn w_all_serving_choice_claims_hold() -> Bool { && w_kv_at_the_context_floor_is_derived_from_architecture() && w_at_the_400k_floor_the_higher_quality_build_is_rejected_on_memory() && w_at_the_400k_floor_the_two_bit_build_is_chosen() - && w_without_the_context_floor_the_higher_quality_build_wins() + && w_the_higher_quality_candidate_wins_when_both_are_admissible() + && w_a_declared_context_never_qualifies_without_a_retrieval_receipt() + && w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() + && w_tied_quality_ranks_refuse_in_both_roster_orders() && w_nothing_admissible_refuses_and_reports_every_rejection() } @@ -184,8 +263,11 @@ data build_unmeasured: QuantizedCandidate = QuantizedCandidate { weights: byte_size(count: 136700000000), quality_rank: 4, declared_context: token_count(count: 1048576), + semantic_context_verified_to: Present { value: token_count(count: 400060) }, + prefill_observations: [ + PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, + ], kv_per_token: none, - measured_prefill_rate: Present { value: tokens_per_second(count: 197) }, } test fn w_an_unmeasured_candidate_is_unanswerable_not_rejected() -> Bool { @@ -193,7 +275,8 @@ test fn w_an_unmeasured_candidate_is_unanswerable_not_rejected() -> Bool { ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: m } => match m { KvFootprintUnmeasured => true - PrefillRateUnmeasured => false + PrefillRateUnmeasuredAtFloor { floor: _ } => false + SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false CapacityMalformed { detail: _ } => false } ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false @@ -210,6 +293,7 @@ test fn w_selection_refuses_while_an_unresolved_candidate_could_win() -> Bool { match c { UnresolvedCandidateCouldWin { unresolved_count: n } => n == 1 CrossReleaseQualityOrderAbsent { distinct_release_count: _ } => false + QualityRankTie { tied_count: _, rank: _ } => false } ChoseCandidate { verdict: _ } => false NoCandidateAdmissible { rejections: _ } => false @@ -224,8 +308,11 @@ data other_release: QuantizedCandidate = QuantizedCandidate { weights: byte_size(count: 22600000000), quality_rank: 3, declared_context: token_count(count: 1048576), + semantic_context_verified_to: Present { value: token_count(count: 400060) }, + prefill_observations: [ + PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 900) }, + ], kv_per_token: Present { value: byte_size(count: 20000) }, - measured_prefill_rate: Present { value: tokens_per_second(count: 900) }, } test fn w_two_admissible_releases_refuse_for_want_of_a_cross_release_order() -> Bool { @@ -235,6 +322,7 @@ test fn w_two_admissible_releases_refuse_for_want_of_a_cross_release_order() -> match c { CrossReleaseQualityOrderAbsent { distinct_release_count: n } => n == 2 UnresolvedCandidateCouldWin { unresolved_count: _ } => false + QualityRankTie { tied_count: _, rank: _ } => false } ChoseCandidate { verdict: _ } => false NoCandidateAdmissible { rejections: _ } => false @@ -260,7 +348,8 @@ test fn w_malformed_capacity_is_unanswerable_not_a_memory_rejection() -> Bool { match m { CapacityMalformed { detail: _ } => true KvFootprintUnmeasured => false - PrefillRateUnmeasured => false + PrefillRateUnmeasuredAtFloor { floor: _ } => false + SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false } ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false @@ -274,3 +363,53 @@ test fn w_all_answerability_claims_hold() -> Bool { && w_two_admissible_releases_refuse_for_want_of_a_cross_release_order() && w_malformed_capacity_is_unanswerable_not_a_memory_rejection() } + +// A TIE HAS NO MAXIMUM. Two admissible candidates of the SAME release at the SAME rank must refuse +// rather than yield to roster order. The second call reverses the roster: a tie-break by position +// would return a different winner for the same inputs, which is why "stable under re-ordering" and +// "keep the earlier one" cannot both be true. +data fixture_tied_a: QuantizedCandidate = QuantizedCandidate { + model_name: "Fixture-Release" as NonEmptyStr, + quant_label: "TIED-A" as NonEmptyStr, + weights: byte_size(count: 20000000000), + quality_rank: 5, + declared_context: token_count(count: 1048576), + semantic_context_verified_to: Present { value: token_count(count: 400060) }, + prefill_observations: [ + PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + ], + kv_per_token: Present { value: byte_size(count: 20000) }, +} + +data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { + model_name: "Fixture-Release" as NonEmptyStr, + quant_label: "TIED-B" as NonEmptyStr, + weights: byte_size(count: 21000000000), + quality_rank: 5, + declared_context: token_count(count: 1048576), + semantic_context_verified_to: Present { value: token_count(count: 400060) }, + prefill_observations: [ + PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + ], + kv_per_token: Present { value: byte_size(count: 20000) }, +} + +fn is_tie_refusal(choice: ServingChoice) -> Bool { + match choice { + SelectionUnanswerable { cause: c, evaluations: _ } => + match c { + QualityRankTie { tied_count: n, rank: r } => n == 2 && r == 5 + UnresolvedCandidateCouldWin { unresolved_count: _ } => false + CrossReleaseQualityOrderAbsent { distinct_release_count: _ } => false + } + ChoseCandidate { verdict: _ } => false + NoCandidateAdmissible { rejections: _ } => false + } +} + +test fn w_tied_quality_ranks_refuse_in_both_roster_orders() -> Bool { + is_tie_refusal(choice: choose_serving_candidate( + candidates: [fixture_tied_a, fixture_tied_b], capacity: node_capacity, constraints: floor_8k)) + && is_tie_refusal(choice: choose_serving_candidate( + candidates: [fixture_tied_b, fixture_tied_a], capacity: node_capacity, constraints: floor_8k)) +} From b825cc388f86c159da02cc9e72288b93d04d6584 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 09:06:34 +0000 Subject: [PATCH 08/22] Key the selector by ReleaseIdentity, and make within-release quality scope structural Side-chat review 5075343704 blocker 4: the cross-release wall was keyed by a model_name string and quality_rank was a bare Int whose scope lived in whichever function remembered not to compare it across releases. A name key is wrong in both directions -- two revisions of one family share a name and merge, one release named two ways splits -- and this PR had already minted an exact ReleaseIdentity for precisely that reason, so counting anything else was a second, weaker key for a fact the identity already decides. QuantizedCandidate and both refusal verdicts now carry it. The ordinal fix is the load-bearing half. WithinReleaseQuality pairs a rank with the release it is ordinal under, compare_within_release returns Absent across releases, and quality_maximum folds through it -- so the cross-release refusal now comes from the comparison having no arm that produces a number, not from the caller checking a distinct count first. That pre-check was validation standing where construction was available (DESIGN 5): satisfiable by editing the caller while the comparison still lied. distinct_release_count survives only to fill the diagnostic. Also drops the `0 - 1` rank verdict_quality returned for non-admissible candidates. That is an invented ordinal: it places a rejected candidate below every real rank on a scale it was never measured on, and collides outright with a release that ranks its own artifacts from -1. Absent instead. release_identity_equal is hoisted to gunbc.model.publication, which owns the identity. population.identity_in had inlined the three-field comparison, so a fourth component added to ReleaseIdentity would have left that consumer silently comparing three. Two things the compiler forced, both of which improve the result: `release` is a RESERVED KEYWORD in every identifier position, not only in a module path, so `release: ReleaseIdentity` refused the whole module index. The field is `identity:`, which is what ModelRelease in publication.dag already calls it. An annotation indented inside a coproduct body is a parse error; annotations are module-item grain. Review 58106 (approving, nit-tier): CapacityMalformed carried a `detail: String`. There is exactly one way a NodeCapacity is malformed, so the string could hold only one value and restated what the constructor says. Payload dropped; the sentence is rendered by missing_fact_wire. Verified by execution, all seven returning true, unfiltered with per-witness rc: w_all_population_claims_hold, w_population_membership_discriminates_in_both_directions, w_all_serving_choice_claims_hold, w_all_answerability_claims_hold, w_tied_quality_ranks_refuse_in_both_roster_orders, w_a_declared_context_never_qualifies_without_a_retrieval_receipt, w_the_higher_quality_candidate_wins_when_both_are_admissible The preceding filtered run of the same set was a FALSE PASS: a module-index refusal prints no line matching "returned|error", so an allowlist grep rendered a tree that did not parse as five clean headers. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01HL72ndT9xg2dZEb6J8a2Gr --- dag/gunbc/model/choice.dag | 200 ++++++++++++------ dag/gunbc/model/population.dag | 8 +- dag/gunbc/model/publication.dag | 11 + .../model/serving_choice_witness_test.dag | 67 ++++-- 4 files changed, 190 insertions(+), 96 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index 05340778ea0..b7d503a06ad 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -2,6 +2,7 @@ module gunbc.model.choice import std.types { String, Bool, List, NonEmptyStr, Int } import std.nat { Nat } +import gunbc.model.publication { ReleaseIdentity, release_identity_equal } import std.measure { ByteSize, byte_size, byte_size_count, TokenCount, token_count, token_count_value, @@ -74,7 +75,7 @@ fn allocatable(capacity: NodeCapacity) -> ByteSize? { // a Q8 of a 30B model are not on one scale. Cross-release quality is the operator's subjective call, // which is why it enters as a floor rather than as an objective. type QuantizedCandidate { - model_name: NonEmptyStr + identity: ReleaseIdentity quant_label: NonEmptyStr weights: ByteSize quality_rank: Int @@ -137,11 +138,15 @@ type ServingConstraints { // A candidate whose KV cost was never measured has not failed the memory test -- it has not TAKEN // it. Reporting that as a rejection would let a fully-measured but worse candidate win by default, // and the answer would look identical to one where the better candidate genuinely lost. +// CapacityMalformed carries NO payload. There is exactly one way a NodeCapacity is malformed -- +// declared runtime overhead exceeding the kernel-visible total -- so the `detail: String` it used to +// carry restated what the constructor already says, and a second malformation would need its own arm +// rather than a different sentence. Prose whose content the type fixes is dead data (DESIGN 4c). type MissingFact = KvFootprintUnmeasured | PrefillRateUnmeasuredAtFloor { floor: TokenCount } | SemanticContextUnverifiedAtFloor { floor: TokenCount, declared: TokenCount } - | CapacityMalformed { detail: String } + | CapacityMalformed fn missing_fact_wire(missing: MissingFact) -> String { match missing { @@ -154,7 +159,7 @@ fn missing_fact_wire(missing: MissingFact) -> String { "; the publisher declares ", to_string(token_count_value(t: d)), " but a declaration is a claim, not evidence", ], "") - CapacityMalformed { detail: d } => join(["capacity-malformed: ", d], "") + CapacityMalformed => "capacity-malformed: declared runtime overhead exceeds the kernel-visible total" } } @@ -168,8 +173,8 @@ type RejectionAxis type CandidateVerdict = ServingCandidateAdmissible { candidate: QuantizedCandidate, resident: ByteSize, headroom: ByteSize } - | ServingCandidateRejected { model_name: NonEmptyStr, quant_label: NonEmptyStr, axis: RejectionAxis } - | ServingCandidateUnanswerable { model_name: NonEmptyStr, quant_label: NonEmptyStr, missing: MissingFact } + | ServingCandidateRejected { identity: ReleaseIdentity, quant_label: NonEmptyStr, axis: RejectionAxis } + | ServingCandidateUnanswerable { identity: ReleaseIdentity, quant_label: NonEmptyStr, missing: MissingFact } // Evaluated in a fixed order so the reported axis is deterministic: capability floors first, then // fit. A candidate that fails several axes reports the first, and the ordering puts the axis the @@ -192,15 +197,13 @@ fn evaluate_candidate( ) -> CandidateVerdict { match allocatable(capacity: capacity) { Absent => ServingCandidateUnanswerable { - model_name: candidate.model_name, + identity: candidate.identity, quant_label: candidate.quant_label, - missing: CapacityMalformed { - detail: "declared runtime overhead exceeds the kernel-visible total, so no allocatable figure exists", - }, + missing: CapacityMalformed, } Present { value: room } => match candidate.kv_per_token { Absent => ServingCandidateUnanswerable { - model_name: candidate.model_name, + identity: candidate.identity, quant_label: candidate.quant_label, missing: KvFootprintUnmeasured, } @@ -213,13 +216,13 @@ fn evaluate_candidate( ) match byte_size_count(b: required) > byte_size_count(b: room) { true => ServingCandidateRejected { - model_name: candidate.model_name, + identity: candidate.identity, quant_label: candidate.quant_label, axis: DoesNotFitMemory { required: required, allocatable: room }, } false => match candidate.semantic_context_verified_to { Absent => ServingCandidateUnanswerable { - model_name: candidate.model_name, + identity: candidate.identity, quant_label: candidate.quant_label, missing: SemanticContextUnverifiedAtFloor { floor: constraints.context_floor, @@ -229,7 +232,7 @@ fn evaluate_candidate( Present { value: verified } => match token_count_value(t: verified) < token_count_value(t: constraints.context_floor) { true => ServingCandidateRejected { - model_name: candidate.model_name, + identity: candidate.identity, quant_label: candidate.quant_label, axis: ContextBelowFloor { declared: verified, floor: constraints.context_floor }, } @@ -238,14 +241,14 @@ fn evaluate_candidate( floor: constraints.context_floor, ) { Absent => ServingCandidateUnanswerable { - model_name: candidate.model_name, + identity: candidate.identity, quant_label: candidate.quant_label, missing: PrefillRateUnmeasuredAtFloor { floor: constraints.context_floor }, } Present { value: prefill } => match tokens_per_second_count(r: prefill) < tokens_per_second_count(r: constraints.prefill_floor) { true => ServingCandidateRejected { - model_name: candidate.model_name, + identity: candidate.identity, quant_label: candidate.quant_label, axis: PrefillBelowFloor { measured: prefill, floor: constraints.prefill_floor }, } @@ -267,16 +270,40 @@ fn evaluate_candidate( fn verdict_is_admissible(verdict: CandidateVerdict) -> Bool { match verdict { ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => true - ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false - ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: _ } => false + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => false } } -fn verdict_quality(verdict: CandidateVerdict) -> Int { +// QUALITY IS ORDINAL WITHIN ONE RELEASE, and this carrier says so structurally. A bare Int rank +// invites the one comparison that has no meaning -- IQ2_XXS of release A against Q4 of release B -- +// and nothing in the type stops it; the guard has to live in whichever function remembers. Pairing +// the rank with the release it is ordinal under moves the guard into `compare_within_release`, the +// single primitive through which every comparison in this module passes. +type WithinReleaseQuality { + identity: ReleaseIdentity + rank: Int +} + +// Absent when the two ranks are ordinal under DIFFERENT releases, which is not a tie and not an +// ordering -- it is the absence of a common scale, and returning any Int here would invent one. +fn compare_within_release(a: WithinReleaseQuality, b: WithinReleaseQuality) -> Int? { + match release_identity_equal(a: a.identity, b: b.identity) { + false => none + true => Present { value: a.rank - b.rank } + } +} + +// Absent for a candidate that is not admissible. The predecessor returned `0 - 1` for those, which +// is an INVENTED ORDINAL: it places a rejected candidate below every real rank on a scale it was +// never measured on, and a release legitimately ranking its own artifacts from -1 would collide with +// it. A fact that was never established is carried as absent, not as a sentinel. +fn verdict_quality(verdict: CandidateVerdict) -> WithinReleaseQuality? { match verdict { - ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => c.quality_rank - ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => 0 - 1 - ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: _ } => 0 - 1 + ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => + Present { value: WithinReleaseQuality { identity: c.identity, rank: c.quality_rank } } + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => none + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => none } } @@ -330,24 +357,24 @@ fn choose_serving_candidate( cause: UnresolvedCandidateCouldWin { unresolved_count: length(unresolved) }, evaluations: verdicts, } - false => match distinct_release_count(verdicts: admissible) > 1 { - true => SelectionUnanswerable { - cause: CrossReleaseQualityOrderAbsent { distinct_release_count: distinct_release_count(verdicts: admissible) }, + false => match quality_maximum(verdicts: admissible) { + NoQualityToCompare => NoCandidateAdmissible { rejections: verdicts } + QualityIncomparableAcrossReleases => SelectionUnanswerable { + cause: CrossReleaseQualityOrderAbsent { + distinct_release_count: distinct_release_count(verdicts: admissible), + }, evaluations: verdicts, } - false => match top_quality_rank(verdicts: admissible) { - Absent => NoCandidateAdmissible { rejections: verdicts } - Present { value: rank } => { - let tied = verdicts_at_rank(verdicts: admissible, rank: rank) - match length(tied) > 1 { - true => SelectionUnanswerable { - cause: QualityRankTie { tied_count: length(tied), rank: rank }, - evaluations: verdicts, - } - false => match first(tied) { - Absent => NoCandidateAdmissible { rejections: verdicts } - Present { value: best } => ChoseCandidate { verdict: best } - } + QualityMaximumFound { quality: top } => { + let tied = verdicts_at_quality(verdicts: admissible, quality: top) + match length(tied) > 1 { + true => SelectionUnanswerable { + cause: QualityRankTie { tied_count: length(tied), rank: top.rank }, + evaluations: verdicts, + } + false => match first(tied) { + Absent => NoCandidateAdmissible { rejections: verdicts } + Present { value: best } => ChoseCandidate { verdict: best } } } } @@ -357,9 +384,9 @@ fn choose_serving_candidate( fn verdict_is_unanswerable(verdict: CandidateVerdict) -> Bool { match verdict { - ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: _ } => true + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => true ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false - ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false } } @@ -370,51 +397,88 @@ fn verdict_is_unanswerable(verdict: CandidateVerdict) -> Bool { // distinct_string_count: std carries no distinct, and reaching into the guarantee-measurement module // for a list primitive would invert the layering -- a model-selection authority would then depend on // a measurement authority for something neither owns. +// How many DISTINCT releases the admissible set spans, keyed by ReleaseIdentity rather than by a +// display name. A name-keyed count is wrong in both directions: two revisions of one family share a +// name and would merge, while one release referred to by an alias in one row and its canonical name +// in another would split. This module already has an exact identity available, so counting anything +// else is a second, weaker key for a fact the identity already decides. +// +// It fills a DIAGNOSTIC and no longer decides anything. The refusal itself now comes from +// `quality_maximum` failing to find a common scale, so the count cannot disagree with the decision. +// +// The dedupe is the corpus's fold/any/append idiom rather than a call to guarantee_measurement's +// distinct_string_count: std carries no distinct, and reaching into the guarantee-measurement module +// for a list primitive would invert the layering -- a model-selection authority would then depend on +// a measurement authority for something neither owns. fn distinct_release_count(verdicts: List) -> Nat { - length(fold(map(verdicts, v => verdict_model_name(verdict: v)), [], (acc, n) => - match any(acc, e => e == n) { + length(fold(map(verdicts, v => verdict_release(verdict: v)), [], (acc, r) => + match any(acc, e => release_identity_equal(a: e, b: r)) { true => acc - false => append(acc, [n]) + false => append(acc, [r]) })) } -fn verdict_model_name(verdict: CandidateVerdict) -> String { +fn verdict_release(verdict: CandidateVerdict) -> ReleaseIdentity { match verdict { - ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => c.model_name as String - ServingCandidateRejected { model_name: m, quant_label: _, axis: _ } => m as String - ServingCandidateUnanswerable { model_name: m, quant_label: _, missing: _ } => m as String + ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => c.identity + ServingCandidateRejected { identity: r, quant_label: _, axis: _ } => r + ServingCandidateUnanswerable { identity: r, quant_label: _, missing: _ } => r } } -// Maximum by quality rank, TOTAL over the empty list because it answers with an option rather than -// a value. The predecessor seeded the fold with a head-or-fabricate helper that manufactured a -// ServingCandidateRejected { model_name: "none", ... DoesNotFitMemory { 0, 0 } } when the list was -// empty. That row was unreachable from the only caller, which is exactly what made it dangerous: it -// is a synthetic verdict a later consumer would read as a real rejection of a real model against a -// real zero-byte capacity, and nothing in the type would mark it as invented. A failure arm must -// refuse rather than fabricate, so emptiness is carried in the RESULT and the caller turns it into -// the honest NoCandidateAdmissible. +// THE MAXIMUM, or a typed statement of why none exists. Three arms because three things are +// genuinely different: nothing to compare, a maximum, and quantities with no common scale. // +// The predecessor returned an Int option and relied on the CALLER checking the release count first. +// That is the arrangement DESIGN 5 names as validation standing where construction was available: +// the guard was satisfiable by editing the caller while the comparison still lied. Here the fold +// cannot produce a number across releases, because `compare_within_release` has no arm that does. +type QualityMaximum + = NoQualityToCompare + | QualityMaximumFound { quality: WithinReleaseQuality } + | QualityIncomparableAcrossReleases + // TIES ARE NOT BROKEN HERE. An earlier revision kept the earlier candidate and its annotation // claimed that made the result stable under re-ordering -- which is exactly backwards, since keeping // the earlier candidate is the one rule whose output DOES change when the roster is reversed. Two // candidates at equal rank have no maximum, and inventing one is the same fabrication this module -// refuses across releases, applied within a release. So this returns the top rank and how many share -// it, and the caller refuses when more than one does. -fn top_quality_rank(verdicts: List) -> Int? { - fold(verdicts, none, (best, v) => - match best { - Absent => Present { value: verdict_quality(verdict: v) } - Present { value: b } => - match verdict_quality(verdict: v) > b { - true => Present { value: verdict_quality(verdict: v) } - false => Present { value: b } +// refuses across releases, applied within a release. So this returns the top quality and the caller +// refuses when more than one candidate shares it. +// +// TOTAL over the empty list, via NoQualityToCompare. The predecessor seeded its fold with a +// head-or-fabricate helper that manufactured a ServingCandidateRejected { "none", DoesNotFitMemory +// { 0, 0 } }. That row was unreachable from its only caller, which is exactly what made it +// dangerous: a synthetic verdict a later consumer would read as a real rejection of a real model +// against a real zero-byte capacity, with nothing in the type marking it invented. +fn quality_maximum(verdicts: List) -> QualityMaximum { + fold(verdicts, NoQualityToCompare, (best, v) => + match verdict_quality(verdict: v) { + Absent => best + Present { value: q } => + match best { + QualityIncomparableAcrossReleases => QualityIncomparableAcrossReleases + NoQualityToCompare => QualityMaximumFound { quality: q } + QualityMaximumFound { quality: b } => + match compare_within_release(a: q, b: b) { + Absent => QualityIncomparableAcrossReleases + Present { value: d } => + match d > 0 { + true => QualityMaximumFound { quality: q } + false => QualityMaximumFound { quality: b } + } + } } }) } -fn verdicts_at_rank(verdicts: List, rank: Int) -> List { - filter(verdicts, v => verdict_quality(verdict: v) == rank) +fn verdicts_at_quality(verdicts: List, quality: WithinReleaseQuality) -> List { + filter(verdicts, v => + match verdict_quality(verdict: v) { + Absent => false + Present { value: q } => + match compare_within_release(a: q, b: quality) { + Absent => false + Present { value: d } => d == 0 + } + }) } - - diff --git a/dag/gunbc/model/population.dag b/dag/gunbc/model/population.dag index f8453a8328f..08065d74140 100644 --- a/dag/gunbc/model/population.dag +++ b/dag/gunbc/model/population.dag @@ -4,7 +4,7 @@ import std.types { String, Bool, List, NonEmptyStr, Int, Timestamp } import std.nat { Nat } import gunbc.model.publication { ModelRelease, ReleaseIdentity, DistributionChannel, DistributionObservation, - release_identity, release_is_open_weight, distribution_channel_wire, + release_identity, release_identity_equal, release_is_open_weight, distribution_channel_wire, } // THE CANDIDATE POPULATION AND ITS NARROWING, built so that the invariant we care about is a @@ -113,11 +113,7 @@ fn narrow_population( } fn identity_in(roster: List, candidate: ReleaseIdentity) -> Bool { - length(filter(roster, i => - (i.publisher as String) == (candidate.publisher as String) - && (i.family as String) == (candidate.family as String) - && (i.revision as String) == (candidate.revision as String) - )) > 0 + length(filter(roster, i => release_identity_equal(a: i, b: candidate))) > 0 } // =========================================================================================== diff --git a/dag/gunbc/model/publication.dag b/dag/gunbc/model/publication.dag index f0e7202543b..e58bba7d692 100644 --- a/dag/gunbc/model/publication.dag +++ b/dag/gunbc/model/publication.dag @@ -103,6 +103,17 @@ type ModelRelease declared_context: DeclaredContext } +// IDENTITY EQUALITY, here rather than at each consumer. A release is identified by publisher, +// family and revision together, so a comparison that omits one silently merges distinct releases -- +// two revisions of one family, or one family published by two organizations. Hoisted from +// population.identity_in, which had inlined the three-field comparison, so a fourth component added +// to ReleaseIdentity cannot leave a consumer comparing the old three. +fn release_identity_equal(a: ReleaseIdentity, b: ReleaseIdentity) -> Bool { + (a.publisher as String) == (b.publisher as String) + && (a.family as String) == (b.family as String) + && (a.revision as String) == (b.revision as String) +} + fn release_identity(subject: ModelRelease) -> ReleaseIdentity { match subject { OpenWeightRelease { identity: i, publication: _, declared_context: _ } => i diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index 4da414a8504..a49f0bfd89f 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -7,6 +7,7 @@ import std.measure { TokenCount, token_count, TokensPerSecond, tokens_per_second, } +import gunbc.model.publication { ReleaseIdentity } import gunbc.model.choice { PrefillRateUnmeasuredAtFloor, SemanticContextUnverifiedAtFloor, PrefillObservation, prefill_rate_at_floor, @@ -39,11 +40,33 @@ data node_capacity: NodeCapacity = NodeCapacity { data kv_per_token: ByteSize = byte_size(count: 88064) -// The two builds actually installed. Quality rank is ordinal WITHIN this release: 3-bit outranks +// RELEASE IDENTITIES, exact rather than display names. Publisher, family and revision together are +// what decide whether two candidates share an ordinal quality scale, so the selector is keyed by +// this and never by a name string: two revisions of one family share a name and would merge under a +// name key, and one release named two ways would split. +data deepseek_v4_flash_release: ReleaseIdentity = ReleaseIdentity { + publisher: "deepseek" as NonEmptyStr, + family: "DeepSeek-V4-Flash" as NonEmptyStr, + revision: "0731" as NonEmptyStr, +} + +data fixture_release: ReleaseIdentity = ReleaseIdentity { + publisher: "fixture-publisher" as NonEmptyStr, + family: "Fixture-Release" as NonEmptyStr, + revision: "1" as NonEmptyStr, +} + +data qwen_release: ReleaseIdentity = ReleaseIdentity { + publisher: "qwen" as NonEmptyStr, + family: "Qwen3.6" as NonEmptyStr, + revision: "35B" as NonEmptyStr, +} + +// The two builds actually installed. Quality rank is ordinal WITHIN this identity: 3-bit outranks // 2-bit. It says nothing about any other model, which is why cross-release quality never enters // this function as a comparable number. data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { - model_name: "DeepSeek-V4-Flash-0731" as NonEmptyStr, + identity: deepseek_v4_flash_release, quant_label: "IQ2_XXS" as NonEmptyStr, weights: byte_size(count: 86720111200), quality_rank: 2, @@ -58,7 +81,7 @@ data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { } data build_iq3_s: QuantizedCandidate = QuantizedCandidate { - model_name: "DeepSeek-V4-Flash-0731" as NonEmptyStr, + identity: deepseek_v4_flash_release, quant_label: "UD-IQ3_S" as NonEmptyStr, weights: byte_size(count: 116100000000), quality_rank: 3, @@ -93,8 +116,8 @@ fn chosen_label(choice: ServingChoice) -> String { ChoseCandidate { verdict: v } => match v { ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => c.quant_label as String - ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => "none" - ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: _ } => "none" + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => "none" + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => "none" } } } @@ -129,14 +152,14 @@ test fn w_kv_at_the_context_floor_is_derived_from_architecture() -> Bool { // says so from the numbers rather than from an argument. test fn w_at_the_400k_floor_the_higher_quality_build_is_rejected_on_memory() -> Bool { match evaluate_candidate(candidate: build_iq3_s, capacity: node_capacity, constraints: floor_400k) { - ServingCandidateRejected { model_name: _, quant_label: _, axis: a } => + ServingCandidateRejected { identity: _, quant_label: _, axis: a } => match a { DoesNotFitMemory { required: _, allocatable: _ } => true ContextBelowFloor { declared: _, floor: _ } => false PrefillBelowFloor { measured: _, floor: _ } => false } ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false - ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: _ } => false + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => false } } @@ -157,7 +180,7 @@ test fn w_at_the_400k_floor_the_two_bit_build_is_chosen() -> Bool { // release, so a maximum over them exists, and they differ only in quality_rank and weights. Without // this control every other claim here is satisfied by a function that returns its first argument. data fixture_low_rank: QuantizedCandidate = QuantizedCandidate { - model_name: "Fixture-Release" as NonEmptyStr, + identity: fixture_release, quant_label: "LOW" as NonEmptyStr, weights: byte_size(count: 20000000000), quality_rank: 1, @@ -170,7 +193,7 @@ data fixture_low_rank: QuantizedCandidate = QuantizedCandidate { } data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { - model_name: "Fixture-Release" as NonEmptyStr, + identity: fixture_release, quant_label: "HIGH" as NonEmptyStr, weights: byte_size(count: 30000000000), quality_rank: 9, @@ -197,16 +220,16 @@ test fn w_the_higher_quality_candidate_wins_when_both_are_admissible() -> Bool { // selector could have quietly reintroduced it. test fn w_a_declared_context_never_qualifies_without_a_retrieval_receipt() -> Bool { match evaluate_candidate(candidate: build_iq3_s, capacity: node_capacity, constraints: floor_8k) { - ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: m } => + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => match m { SemanticContextUnverifiedAtFloor { floor: _, declared: d } => token_count_value(t: d) == 1048576 KvFootprintUnmeasured => false PrefillRateUnmeasuredAtFloor { floor: _ } => false - CapacityMalformed { detail: _ } => false + CapacityMalformed => false } ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false - ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false } } @@ -258,7 +281,7 @@ test fn w_all_serving_choice_claims_hold() -> Bool { // A candidate whose KV cost was never measured has not FAILED the memory test -- it has not taken // it. The selector must refuse rather than hand the win to the measured candidate by default. data build_unmeasured: QuantizedCandidate = QuantizedCandidate { - model_name: "DeepSeek-V4-Flash-0731" as NonEmptyStr, + identity: deepseek_v4_flash_release, quant_label: "UD-IQ4_XS" as NonEmptyStr, weights: byte_size(count: 136700000000), quality_rank: 4, @@ -272,14 +295,14 @@ data build_unmeasured: QuantizedCandidate = QuantizedCandidate { test fn w_an_unmeasured_candidate_is_unanswerable_not_rejected() -> Bool { match evaluate_candidate(candidate: build_unmeasured, capacity: node_capacity, constraints: floor_400k) { - ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: m } => + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => match m { KvFootprintUnmeasured => true PrefillRateUnmeasuredAtFloor { floor: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false - CapacityMalformed { detail: _ } => false + CapacityMalformed => false } - ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false } } @@ -303,7 +326,7 @@ test fn w_selection_refuses_while_an_unresolved_candidate_could_win() -> Bool { // Preference is ordinal WITHIN a release, so two admissible releases have no join and a maximum over // them does not exist. Returning one would manufacture an ordering the inputs never contained. data other_release: QuantizedCandidate = QuantizedCandidate { - model_name: "Qwen3.6-35B" as NonEmptyStr, + identity: qwen_release, quant_label: "Q8_0" as NonEmptyStr, weights: byte_size(count: 22600000000), quality_rank: 3, @@ -344,14 +367,14 @@ test fn w_malformed_capacity_is_unanswerable_not_a_memory_rejection() -> Bool { Present { value: _ } => false Absent => match evaluate_candidate(candidate: build_iq2_xxs, capacity: malformed_capacity, constraints: floor_400k) { - ServingCandidateUnanswerable { model_name: _, quant_label: _, missing: m } => + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => match m { - CapacityMalformed { detail: _ } => true + CapacityMalformed => true KvFootprintUnmeasured => false PrefillRateUnmeasuredAtFloor { floor: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false } - ServingCandidateRejected { model_name: _, quant_label: _, axis: _ } => false + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false } } @@ -369,7 +392,7 @@ test fn w_all_answerability_claims_hold() -> Bool { // would return a different winner for the same inputs, which is why "stable under re-ordering" and // "keep the earlier one" cannot both be true. data fixture_tied_a: QuantizedCandidate = QuantizedCandidate { - model_name: "Fixture-Release" as NonEmptyStr, + identity: fixture_release, quant_label: "TIED-A" as NonEmptyStr, weights: byte_size(count: 20000000000), quality_rank: 5, @@ -382,7 +405,7 @@ data fixture_tied_a: QuantizedCandidate = QuantizedCandidate { } data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { - model_name: "Fixture-Release" as NonEmptyStr, + identity: fixture_release, quant_label: "TIED-B" as NonEmptyStr, weights: byte_size(count: 21000000000), quality_rank: 5, From 911b4a3c7462cf111cc607c2036468ba6ad8d393 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 09:34:18 +0000 Subject: [PATCH 09/22] Regime mismatch is unanswerable, not satisfied by whatever was measured Side-chat blocker 2, to the bar they set: the selector must make regime mismatch unanswerable rather than let a fresh-prefill receipt satisfy a warm constraint. ServingRegime = FreshSessionPrefill | WarmContinuation | RestoredSession is now a field on PrefillObservation and on ServingConstraints, and prefill_rate_at_floor filters on regime equality as well as depth. There is no widening arm: a candidate holding only fresh receipts, asked about a warm floor, returns Absent, which evaluate_candidate turns into PrefillRateUnmeasuredAtFloor { floor, regime } and so into Unanswerable. It has not failed the test, it has not taken it. The substitution is flattering in exactly one direction, which is why the arm cannot exist rather than merely being unused: warm numbers are the fast ones, so letting a warm receipt answer a fresh question admits a realization that cannot serve a cold request at the floor. This fleet measures 2963 s for a cold 400k prefill against seconds for an appended turn -- three orders of magnitude, not measurement noise. Every observation in the fixture is labelled FreshSessionPrefill, which is what they are: every latency number measured on the nodes was a cold prefill. RestoredSession is enumerated with no evidence behind it deliberately, so a future observation of a session reloaded after eviction or slot reassignment must say which of the three it is rather than defaulting into WarmContinuation. Two witnesses, on the helper and on the selector, so the refusal is not merely a property of a filter that never fires. The first carries its own positive control: the SAME observation list does answer at FreshSessionPrefill and yields 135 tok/s, so the Absent is not satisfied by a filter that rejects everything. NOT addressed here, and still owed: the reuse-evidence receipt (same realization and cache identity, known reused-token population) and the request-concurrency condition. Those are observation-side obligations a selector cannot discharge; this change only ensures a warm claim cannot be made without them. Review 58113 (approving, nit-tier): distinct_release_count carried two doc blocks -- the ReleaseIdentity rewrite left the pre-rekey annotation stranded above its replacement, so one explained a name key and one an identity key. Stale one deleted. Verified by execution, all seven returning true, unfiltered with per-witness rc: w_all_serving_choice_claims_hold, w_all_answerability_claims_hold, w_a_fresh_prefill_receipt_does_not_answer_a_warm_continuation_floor, w_a_warm_floor_over_fresh_only_receipts_is_unanswerable, w_tied_quality_ranks_refuse_in_both_roster_orders, w_all_population_claims_hold, w_population_membership_discriminates_in_both_directions Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01HL72ndT9xg2dZEb6J8a2Gr --- dag/gunbc/model/choice.dag | 75 ++++++++++++---- .../model/serving_choice_witness_test.dag | 88 ++++++++++++++++--- 2 files changed, 137 insertions(+), 26 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index b7d503a06ad..99039380df9 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -90,15 +90,59 @@ type QuantizedCandidate { // 135 at 400,060: attention cost grows with depth, so a single rate silently describes whichever // depth the measurer happened to use. A consumer asking "is this fast enough at my floor" needs an // observation AT OR BEYOND that floor; anything shallower flatters the candidate. +// THE OPERATING REGIME A LATENCY NUMBER WAS MEASURED UNDER. Fresh prefill, warm continuation and a +// restored session are not the same quantity measured with different luck -- they differ by three +// orders of magnitude on this fleet, because a fresh 400k prefill pays for every token while a warm +// continuation pays only for the appended turn. A rate carried without its regime therefore silently +// answers a question it never measured, and the direction of the error is always flattering: warm +// numbers are the fast ones, so an unlabelled roster drifts toward claiming warm performance for +// cold requests. +// +// RestoredSession is enumerated but deliberately has no evidence behind it yet. It is the parked +// case -- state reloaded after eviction or slot reassignment -- and naming it here means a future +// observation must say which of the three it is rather than defaulting into WarmContinuation. +type ServingRegime = FreshSessionPrefill | WarmContinuation | RestoredSession + +fn serving_regime_wire(regime: ServingRegime) -> String { + match regime { + FreshSessionPrefill => "fresh-session-prefill" + WarmContinuation => "warm-continuation" + RestoredSession => "restored-session" + } +} + +fn same_regime(a: ServingRegime, b: ServingRegime) -> Bool { + match a { + FreshSessionPrefill => match b { FreshSessionPrefill => true WarmContinuation => false RestoredSession => false } + WarmContinuation => match b { FreshSessionPrefill => false WarmContinuation => true RestoredSession => false } + RestoredSession => match b { FreshSessionPrefill => false WarmContinuation => false RestoredSession => true } + } +} + type PrefillObservation { + regime: ServingRegime depth: TokenCount rate: TokensPerSecond } -// The slowest rate observed at or beyond the floor, or Absent when nothing reaches that depth. -// Slowest rather than fastest because a floor is a promise about the worst case a caller will meet. -fn prefill_rate_at_floor(observations: List, floor: TokenCount) -> TokensPerSecond? { - let deep = filter(observations, o => token_count_value(t: o.depth) >= token_count_value(t: floor)) +// The slowest rate observed AT THE REQUESTED REGIME at or beyond the floor, or Absent when no such +// observation exists. Slowest rather than fastest because a floor is a promise about the worst case +// a caller will meet. +// +// REGIME MISMATCH IS ABSENT, NOT A FALLBACK TO WHATEVER WAS MEASURED. A candidate holding only +// fresh-prefill observations, asked about a warm-continuation floor, is UNANSWERABLE -- it has not +// failed the test, it has not taken it. Substituting the fresh number would refuse a candidate that +// may serve warm perfectly well; substituting a warm number for a fresh question is the dangerous +// direction and would admit one that cannot. Neither substitution is available here because the +// filter is on regime equality, so the widening arm does not exist to be taken. +fn prefill_rate_at_floor( + observations: List, + floor: TokenCount, + regime: ServingRegime, +) -> TokensPerSecond? { + let deep = filter(observations, o => + same_regime(a: o.regime, b: regime) + && token_count_value(t: o.depth) >= token_count_value(t: floor)) fold(deep, none, (worst, o) => match worst { Absent => Present { value: o.rate } @@ -130,6 +174,7 @@ type ServingConstraints { context_floor: TokenCount hot_sessions: Nat prefill_floor: TokensPerSecond + prefill_regime: ServingRegime } // ============================ THE DECISION ============================ @@ -144,15 +189,18 @@ type ServingConstraints { // rather than a different sentence. Prose whose content the type fixes is dead data (DESIGN 4c). type MissingFact = KvFootprintUnmeasured - | PrefillRateUnmeasuredAtFloor { floor: TokenCount } + | PrefillRateUnmeasuredAtFloor { floor: TokenCount, regime: ServingRegime } | SemanticContextUnverifiedAtFloor { floor: TokenCount, declared: TokenCount } | CapacityMalformed fn missing_fact_wire(missing: MissingFact) -> String { match missing { KvFootprintUnmeasured => "kv-footprint-unmeasured" - PrefillRateUnmeasuredAtFloor { floor: f } => - join(["no prefill observation at or beyond ", to_string(token_count_value(t: f))], "") + PrefillRateUnmeasuredAtFloor { floor: f, regime: g } => + join([ + "no ", serving_regime_wire(regime: g), " observation at or beyond ", + to_string(token_count_value(t: f)), + ], "") SemanticContextUnverifiedAtFloor { floor: f, declared: d } => join([ "retrieval unverified at ", to_string(token_count_value(t: f)), @@ -239,11 +287,15 @@ fn evaluate_candidate( false => match prefill_rate_at_floor( observations: candidate.prefill_observations, floor: constraints.context_floor, + regime: constraints.prefill_regime, ) { Absent => ServingCandidateUnanswerable { identity: candidate.identity, quant_label: candidate.quant_label, - missing: PrefillRateUnmeasuredAtFloor { floor: constraints.context_floor }, + missing: PrefillRateUnmeasuredAtFloor { + floor: constraints.context_floor, + regime: constraints.prefill_regime, + }, } Present { value: prefill } => match tokens_per_second_count(r: prefill) < tokens_per_second_count(r: constraints.prefill_floor) { @@ -390,13 +442,6 @@ fn verdict_is_unanswerable(verdict: CandidateVerdict) -> Bool { } } -// How many DISTINCT releases the admissible set spans. Preference is ordinal within one release, so -// more than one release means the maximum this function would return does not exist. -// -// The dedupe is the corpus's fold/any/append idiom rather than a call to guarantee_measurement's -// distinct_string_count: std carries no distinct, and reaching into the guarantee-measurement module -// for a list primitive would invert the layering -- a model-selection authority would then depend on -// a measurement authority for something neither owns. // How many DISTINCT releases the admissible set spans, keyed by ReleaseIdentity rather than by a // display name. A name-keyed count is wrong in both directions: two revisions of one family share a // name and would merge, while one release referred to by an alias in one row and its canonical name diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index a49f0bfd89f..e9fa31a8841 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -5,12 +5,13 @@ import std.nat { Nat } import std.measure { ByteSize, byte_size, byte_size_count, TokenCount, token_count, - TokensPerSecond, tokens_per_second, + TokensPerSecond, tokens_per_second, tokens_per_second_count, } import gunbc.model.publication { ReleaseIdentity } import gunbc.model.choice { PrefillRateUnmeasuredAtFloor, SemanticContextUnverifiedAtFloor, PrefillObservation, prefill_rate_at_floor, + FreshSessionPrefill, WarmContinuation, serving_regime_wire, NodeCapacity, QuantizedCandidate, ServingConstraints, CandidateVerdict, ServingCandidateAdmissible, ServingCandidateRejected, RejectionAxis, DoesNotFitMemory, ContextBelowFloor, PrefillBelowFloor, @@ -73,9 +74,9 @@ data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, prefill_observations: [ - PrefillObservation { depth: token_count(count: 160060), rate: tokens_per_second(count: 253) }, - PrefillObservation { depth: token_count(count: 255061), rate: tokens_per_second(count: 197) }, - PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, + PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 160060), rate: tokens_per_second(count: 253) }, + PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 255061), rate: tokens_per_second(count: 197) }, + PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, ], kv_per_token: Present { value: kv_per_token }, } @@ -101,12 +102,14 @@ data floor_400k: ServingConstraints = ServingConstraints { context_floor: token_count(count: 400000), hot_sessions: 1, prefill_floor: tokens_per_second(count: 0), + prefill_regime: FreshSessionPrefill, } data floor_8k: ServingConstraints = ServingConstraints { context_floor: token_count(count: 8192), hot_sessions: 1, prefill_floor: tokens_per_second(count: 0), + prefill_regime: FreshSessionPrefill, } fn chosen_label(choice: ServingChoice) -> String { @@ -187,7 +190,7 @@ data fixture_low_rank: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, prefill_observations: [ - PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], kv_per_token: Present { value: byte_size(count: 20000) }, } @@ -200,7 +203,7 @@ data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, prefill_observations: [ - PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 400) }, + PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 400) }, ], kv_per_token: Present { value: byte_size(count: 20000) }, } @@ -237,8 +240,13 @@ test fn w_a_declared_context_never_qualifies_without_a_retrieval_receipt() -> Bo // (253 -> 197 -> 135 measured on one realization), so a shallow reading flatters the candidate. test fn w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() -> Bool { match prefill_rate_at_floor( - observations: [PrefillObservation { depth: token_count(count: 160060), rate: tokens_per_second(count: 253) }], + observations: [PrefillObservation { + regime: FreshSessionPrefill, + depth: token_count(count: 160060), + rate: tokens_per_second(count: 253), + }], floor: token_count(count: 400000), + regime: FreshSessionPrefill, ) { Absent => true Present { value: _ } => false @@ -253,6 +261,7 @@ test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { context_floor: token_count(count: 8000000), hot_sessions: 1, prefill_floor: tokens_per_second(count: 0), + prefill_regime: FreshSessionPrefill, } match choose_serving_candidate( candidates: installed(), capacity: node_capacity, constraints: impossible) { @@ -262,6 +271,61 @@ test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { } } +// Carries its own POSITIVE CONTROL in the second arm: the identical observation list DOES answer at +// the regime it was actually measured under, so the Absent is not satisfied by a filter that rejects +// everything. +// +// REGIME MISMATCH IS UNANSWERABLE, NOT SATISFIED. Every observation this repository holds is a +// FRESH prefill measurement. Asked about a WARM continuation floor, a candidate carrying only fresh +// receipts has not failed -- it has not been measured, and the selector must say so. +// +// This is the arm that matters commercially, because the substitution is flattering in exactly one +// direction: warm numbers are the fast ones, so letting a warm receipt answer a fresh question would +// admit a realization that cannot serve a cold request at the floor. Here neither substitution is +// available, because the filter is regime equality and there is no widening arm to take. +test fn w_a_fresh_prefill_receipt_does_not_answer_a_warm_continuation_floor() -> Bool { + let deep_fresh = [PrefillObservation { + regime: FreshSessionPrefill, + depth: token_count(count: 400060), + rate: tokens_per_second(count: 135), + }] + match prefill_rate_at_floor( + observations: deep_fresh, floor: token_count(count: 400000), regime: WarmContinuation, + ) { + Present { value: _ } => false + Absent => + match prefill_rate_at_floor( + observations: deep_fresh, floor: token_count(count: 400000), regime: FreshSessionPrefill, + ) { + Absent => false + Present { value: r } => tokens_per_second_count(r: r) == 135 + } + } +} + +// And the same mismatch reaching the SELECTOR, so the refusal is not merely a property of the helper. +// A warm floor over the installed builds is Unanswerable, and it names the regime it lacked. +test fn w_a_warm_floor_over_fresh_only_receipts_is_unanswerable() -> Bool { + let warm_400k = ServingConstraints { + context_floor: token_count(count: 400000), + hot_sessions: 1, + prefill_floor: tokens_per_second(count: 1), + prefill_regime: WarmContinuation, + } + match evaluate_candidate(candidate: build_iq2_xxs, capacity: node_capacity, constraints: warm_400k) { + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => + match m { + PrefillRateUnmeasuredAtFloor { floor: _, regime: g } => + serving_regime_wire(regime: g) == "warm-continuation" + KvFootprintUnmeasured => false + SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false + CapacityMalformed => false + } + ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false + } +} + test fn w_all_serving_choice_claims_hold() -> Bool { w_capacity_grains_differ_and_only_kernel_visible_is_spendable() && w_kv_at_the_context_floor_is_derived_from_architecture() @@ -272,6 +336,8 @@ test fn w_all_serving_choice_claims_hold() -> Bool { && w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() && w_tied_quality_ranks_refuse_in_both_roster_orders() && w_nothing_admissible_refuses_and_reports_every_rejection() + && w_a_fresh_prefill_receipt_does_not_answer_a_warm_continuation_floor() + && w_a_warm_floor_over_fresh_only_receipts_is_unanswerable() } // ======================= THE ANSWERABILITY CLAIMS ======================= @@ -288,7 +354,7 @@ data build_unmeasured: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, prefill_observations: [ - PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, + PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, ], kv_per_token: none, } @@ -333,7 +399,7 @@ data other_release: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, prefill_observations: [ - PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 900) }, + PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 900) }, ], kv_per_token: Present { value: byte_size(count: 20000) }, } @@ -399,7 +465,7 @@ data fixture_tied_a: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, prefill_observations: [ - PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], kv_per_token: Present { value: byte_size(count: 20000) }, } @@ -412,7 +478,7 @@ data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, prefill_observations: [ - PrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], kv_per_token: Present { value: byte_size(count: 20000) }, } From 14fb18ab0ca6ff2fb649496674d7853328257c30 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 10:44:10 +0000 Subject: [PATCH 10/22] A warm rate is unwritable, the population is release-grained, and the buy-nodes rule is gone Three narrow fixes, each closing a gap the reviewer located on 911b4a3c746. REGIME PROVENANCE. Discrimination was solved; provenance was not. `PrefillObservation` carried a three-arm regime field, so `{ regime: WarmContinuation, rate: 500 }` was an ordinary writable record -- the fast number, mintable by anyone, with no reuse receipt behind it. The carrier is now `FreshPrefillObservation` with no regime field at all: the only regime whose provenance needs no receipt is the one that measures itself. Warm and restored floors return `none` structurally rather than by filtering, so the refusal survives any edit short of building the receipt, whose carrier is named as the trigger. POPULATION GRAIN. `PopulationStage` enumerated five realization verdicts over members typed `List`. This repository's own measurement falsifies that keying: DeepSeek V4 Flash 0731 at IQ2_XXS is single-node feasible at a 400k floor and the same release at IQ3_S is not, and a release-keyed stage answers "DeepSeek V4 is in" for both. The stages are removed rather than renamed, with RealizationIdentity/RealizationPopulation named as the carrier they belong to. RED 3 now witnesses the stronger invariant it was always reaching for: no narrowing, empty or otherwise, reaches back into its parent. STALE DERIVATION. Two sites still said a memory-bound census means more nodes help. That rule was withdrawn: a memory rejection proves only that more USABLE memory would remove it, and whether another node supplies memory this realization can use is a distributed- runtime and topology fact the axis does not carry. The hardware question is the counterfactual -- choose(current) against choose(current + proposed) -- and a rejection census is its input, not its answer. Every witness re-run and passing at identity grain: population, serving choice, answerability, and the tie refusal. --- dag/gunbc/model/choice.dag | 76 ++++++++++++------- dag/gunbc/model/population.dag | 33 ++++---- ...odel_population_narrowing_witness_test.dag | 18 +++-- .../model/serving_choice_witness_test.dag | 46 +++++------ 4 files changed, 103 insertions(+), 70 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index 99039380df9..77252d080ba 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -24,8 +24,16 @@ import std.measure { // // WHAT IT REFUSES TO DO: pick silently when nothing fits. A refusal names the BINDING AXIS per // candidate, which is what makes "would more hardware help" a derivable question instead of a -// debate. If every candidate is bound by memory, more nodes help; if they are bound by a context -// floor no published artifact reaches, more nodes buy nothing and the floor is the thing to move. +// debate. +// +// IT IS NOT, HOWEVER, AN ANSWER TO THAT QUESTION. A memory rejection proves only that more USABLE +// memory would remove this rejection -- not that another node supplies memory this realization can +// use, which depends on distributed-runtime and topology facts this axis does not carry. The +// hardware question is the counterfactual, and it is answered by running this function twice: +// +// choose(current_supply) versus choose(current_supply + proposed_supply) +// +// A rejection census is the INPUT to that comparison, never a substitute for it. // ============================ CAPACITY, at the three grains that actually differ ============================ @@ -81,7 +89,7 @@ type QuantizedCandidate { quality_rank: Int declared_context: TokenCount semantic_context_verified_to: TokenCount? - prefill_observations: List + fresh_prefill_observations: List kv_per_token: ByteSize? } @@ -111,16 +119,23 @@ fn serving_regime_wire(regime: ServingRegime) -> String { } } -fn same_regime(a: ServingRegime, b: ServingRegime) -> Bool { - match a { - FreshSessionPrefill => match b { FreshSessionPrefill => true WarmContinuation => false RestoredSession => false } - WarmContinuation => match b { FreshSessionPrefill => false WarmContinuation => true RestoredSession => false } - RestoredSession => match b { FreshSessionPrefill => false WarmContinuation => false RestoredSession => true } - } -} - -type PrefillObservation { - regime: ServingRegime +// AN OBSERVATION CARRIES NO REGIME FIELD, and that absence is the wall. +// +// Enumerating three regimes on the observation would make `{ regime: WarmContinuation, rate: 500 }` +// an ordinary writable record: the fast number, mintable by anyone, with nothing behind it. The +// discrimination would be real and the PROVENANCE would be fiction, which is the flattering +// direction of the error this split exists to close. +// +// A warm or restored rate is only a fact about serving if something attests the reuse it claims -- +// the same realization, a cache/session identity, the reused-token population, the concurrency +// condition it held under. This repository can produce none of those today. So the only regime an +// observation can be about is the one whose provenance needs no receipt: a fresh prefill measures +// itself. Warm and restored floors are therefore UNANSWERABLE rather than answered, and no edit to +// this module short of building the receipt can change that. +// +// NEXT-RUNG TRIGGER: a PrefixReuseReceipt carrier produced by the observation layer, at which point +// a WarmPrefillObservation may be constructed FROM one and not otherwise. +type FreshPrefillObservation { depth: TokenCount rate: TokensPerSecond } @@ -129,20 +144,29 @@ type PrefillObservation { // observation exists. Slowest rather than fastest because a floor is a promise about the worst case // a caller will meet. // -// REGIME MISMATCH IS ABSENT, NOT A FALLBACK TO WHATEVER WAS MEASURED. A candidate holding only -// fresh-prefill observations, asked about a warm-continuation floor, is UNANSWERABLE -- it has not -// failed the test, it has not taken it. Substituting the fresh number would refuse a candidate that -// may serve warm perfectly well; substituting a warm number for a fresh question is the dangerous -// direction and would admit one that cannot. Neither substitution is available here because the -// filter is on regime equality, so the widening arm does not exist to be taken. +// REGIME MISMATCH IS ABSENT, NOT A FALLBACK TO WHATEVER WAS MEASURED. A candidate asked about a +// warm-continuation floor is UNANSWERABLE -- it has not failed the test, it has not taken it. +// Substituting the fresh number would refuse a candidate that may serve warm perfectly well; +// substituting a warm number for a fresh question is the dangerous direction and would admit one +// that cannot. Neither substitution is available here: the non-fresh arms have no observation to +// read at all, so the widening arm does not exist to be taken. fn prefill_rate_at_floor( - observations: List, + observations: List, floor: TokenCount, regime: ServingRegime, ) -> TokensPerSecond? { - let deep = filter(observations, o => - same_regime(a: o.regime, b: regime) - && token_count_value(t: o.depth) >= token_count_value(t: floor)) + match regime { + WarmContinuation => none + RestoredSession => none + FreshSessionPrefill => slowest_fresh_rate_at_floor(observations: observations, floor: floor) + } +} + +fn slowest_fresh_rate_at_floor( + observations: List, + floor: TokenCount, +) -> TokensPerSecond? { + let deep = filter(observations, o => token_count_value(t: o.depth) >= token_count_value(t: floor)) fold(deep, none, (worst, o) => match worst { Absent => Present { value: o.rate } @@ -212,8 +236,8 @@ fn missing_fact_wire(missing: MissingFact) -> String { } // Why a candidate was rejected, at the grain that tells you what to change. One axis per arm, so a -// census over rejections partitions cleanly into "buy hardware", "lower a floor", "wait for a -// better artifact". +// census over rejections says which INPUT would have to move -- usable memory, a declared floor, a +// published artifact -- without claiming which purchase moves it. type RejectionAxis = DoesNotFitMemory { required: ByteSize, allocatable: ByteSize } | ContextBelowFloor { declared: TokenCount, floor: TokenCount } @@ -285,7 +309,7 @@ fn evaluate_candidate( axis: ContextBelowFloor { declared: verified, floor: constraints.context_floor }, } false => match prefill_rate_at_floor( - observations: candidate.prefill_observations, + observations: candidate.fresh_prefill_observations, floor: constraints.context_floor, regime: constraints.prefill_regime, ) { diff --git a/dag/gunbc/model/population.dag b/dag/gunbc/model/population.dag index 08065d74140..801bbfb1766 100644 --- a/dag/gunbc/model/population.dag +++ b/dag/gunbc/model/population.dag @@ -12,8 +12,9 @@ import gunbc.model.publication { // // THE INVARIANT: failure to enter a later population must never remove membership from an earlier // one. Open weights exist regardless of whether a packager packaged them, a runtime loads them, a -// host fits them, or they answer well. Those are downstream verdicts about a REALIZATION; none is a -// fact about the release. +// host fits them, or they answer well. Whether any packager packaged the release is still a fact +// about the release; whether a particular artifact loads, fits or answers well is a verdict about a +// REALIZATION, and no such verdict is representable at this grain -- see PopulationStage. // // HOW IT IS MADE STRUCTURAL RATHER THAN CHECKED. A narrowed population is not authored -- it is // CONSTRUCTED FROM A PARENT by filtering, and it retains the parent. Members of a stage are a subset @@ -23,26 +24,28 @@ import gunbc.model.publication { // regression to be written into. A lens checking this afterwards would be a second representation of // something the construction already guarantees. -// The ordered stages. Each names what it is a verdict about, and every one below the first is a -// verdict about a REALIZATION of the release rather than about the release. +// The ordered stages. Each names what it is a verdict about, and BOTH are verdicts about the +// RELEASE -- which is the grain this population's members are keyed at. +// +// WHY THE LATER STAGES ARE NOT HERE. Runtime compatibility, single- and multi-node feasibility, +// context qualification and coding qualification are verdicts about a REALIZATION -- one artifact, +// one quantization, one runtime -- and this repository's own measurements falsify keying them by +// release: DeepSeek V4 Flash 0731 at IQ2_XXS is single-node feasible at a 400k floor while the SAME +// release at IQ3_S is not. A `List` cannot retain that distinction, so a stage keyed +// here would answer "DeepSeek V4 is in" for both and lose the only fact the measurement produced. +// Enumerating them anyway would be rung inflation: a stage name that reads as a modeled verdict +// while the carrier underneath cannot hold the verdict's subject. +// +// NEXT-RUNG TRIGGER: a RealizationIdentity distinguishing artifact x quantization x runtime, and a +// RealizationPopulation whose members carry it. Those stages belong to that carrier, not this one. type PopulationStage = OpenWeightStage | LocallyPackagedStage - | RuntimeCompatibleStage - | SingleNodeFeasibleStage - | MultiNodeFeasibleStage - | ContextQualifiedStage - | CodingQualifiedStage fn population_stage_wire(stage: PopulationStage) -> String { match stage { OpenWeightStage => "open-weight" LocallyPackagedStage => "locally-packaged" - RuntimeCompatibleStage => "runtime-compatible" - SingleNodeFeasibleStage => "single-node-feasible" - MultiNodeFeasibleStage => "multi-node-feasible" - ContextQualifiedStage => "context-qualified" - CodingQualifiedStage => "coding-qualified" } } @@ -135,7 +138,7 @@ fn open_weight_completeness(population: ModelPopulation) -> CompletenessVerdict "completeness about the open-weight universe was asked of the ", population_stage_wire(stage: population.stage), " stage, which is a narrowing of ", population_stage_wire(stage: p), - " -- a later stage is a verdict about realizations and under-reports the universe", + " -- a narrowed stage is a filtered subset and under-reports the universe", ], ""), } DiscoveredRoot { sources: sources } => diff --git a/dag/test/claim/model/model_population_narrowing_witness_test.dag b/dag/test/claim/model/model_population_narrowing_witness_test.dag index e45a14b0b9e..f6ce1fc9512 100644 --- a/dag/test/claim/model/model_population_narrowing_witness_test.dag +++ b/dag/test/claim/model/model_population_narrowing_witness_test.dag @@ -14,7 +14,7 @@ import gunbc.model.publication { } import gunbc.model.population { ModelPopulation, - LocallyPackagedStage, SingleNodeFeasibleStage, + LocallyPackagedStage, ReleaseDiscoverySource, PublisherReleaseCatalog, SingleDistributorCatalog, open_weight_population, narrow_population, @@ -122,10 +122,16 @@ test fn w_ollama_cloud_presence_does_not_imply_local_weights_are_unavailable() - && population_holds(population: root(), id: deepseek_v4_flash_id) } -// RED 3 -- not fitting one host was read as the model being unavailable. -test fn w_single_node_infeasibility_does_not_remove_from_the_open_weight_population() -> Bool { - let feasible = narrow_population(parent: root(), stage: SingleNodeFeasibleStage, admits: []) - length(feasible.members) == 0 +// RED 3 -- a narrowing that admits NOTHING must still leave the root intact. This is the general +// form of the original error: not fitting one host was read as the model being unavailable. +// +// The specific single-node verdict is no longer expressible as a population stage, because it is a +// verdict about a REALIZATION and this population is keyed by release -- see PopulationStage. The +// invariant it was witnessing is the one asserted here, and it is the stronger statement: NO +// narrowing, empty or otherwise, reaches back into its parent. +test fn w_an_empty_narrowing_does_not_remove_from_the_open_weight_population() -> Bool { + let admitted_nothing = narrow_population(parent: root(), stage: LocallyPackagedStage, admits: []) + length(admitted_nothing.members) == 0 && population_holds(population: root(), id: deepseek_v4_flash_id) } @@ -188,7 +194,7 @@ test fn w_a_narrowed_stage_refuses_completeness_about_the_open_weight_universe() test fn w_all_population_claims_hold() -> Bool { w_ollama_local_absence_does_not_remove_from_the_open_weight_population() && w_ollama_cloud_presence_does_not_imply_local_weights_are_unavailable() - && w_single_node_infeasibility_does_not_remove_from_the_open_weight_population() + && w_an_empty_narrowing_does_not_remove_from_the_open_weight_population() && w_narrowing_cannot_introduce_a_member_the_parent_lacked() && w_a_closed_weight_release_is_excluded_from_the_open_weight_population() && w_completeness_refuses_when_every_source_is_a_single_distributor() diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index e9fa31a8841..acc6e256a95 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -10,7 +10,7 @@ import std.measure { import gunbc.model.publication { ReleaseIdentity } import gunbc.model.choice { PrefillRateUnmeasuredAtFloor, SemanticContextUnverifiedAtFloor, - PrefillObservation, prefill_rate_at_floor, + FreshPrefillObservation, prefill_rate_at_floor, FreshSessionPrefill, WarmContinuation, serving_regime_wire, NodeCapacity, QuantizedCandidate, ServingConstraints, CandidateVerdict, ServingCandidateAdmissible, ServingCandidateRejected, @@ -73,10 +73,10 @@ data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { quality_rank: 2, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, - prefill_observations: [ - PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 160060), rate: tokens_per_second(count: 253) }, - PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 255061), rate: tokens_per_second(count: 197) }, - PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, + fresh_prefill_observations: [ + FreshPrefillObservation { depth: token_count(count: 160060), rate: tokens_per_second(count: 253) }, + FreshPrefillObservation { depth: token_count(count: 255061), rate: tokens_per_second(count: 197) }, + FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, ], kv_per_token: Present { value: kv_per_token }, } @@ -88,7 +88,7 @@ data build_iq3_s: QuantizedCandidate = QuantizedCandidate { quality_rank: 3, declared_context: token_count(count: 1048576), semantic_context_verified_to: none, - prefill_observations: [], + fresh_prefill_observations: [], kv_per_token: Present { value: kv_per_token }, } @@ -189,8 +189,8 @@ data fixture_low_rank: QuantizedCandidate = QuantizedCandidate { quality_rank: 1, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, - prefill_observations: [ - PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + fresh_prefill_observations: [ + FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], kv_per_token: Present { value: byte_size(count: 20000) }, } @@ -202,8 +202,8 @@ data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { quality_rank: 9, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, - prefill_observations: [ - PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 400) }, + fresh_prefill_observations: [ + FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 400) }, ], kv_per_token: Present { value: byte_size(count: 20000) }, } @@ -240,8 +240,7 @@ test fn w_a_declared_context_never_qualifies_without_a_retrieval_receipt() -> Bo // (253 -> 197 -> 135 measured on one realization), so a shallow reading flatters the candidate. test fn w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() -> Bool { match prefill_rate_at_floor( - observations: [PrefillObservation { - regime: FreshSessionPrefill, + observations: [FreshPrefillObservation { depth: token_count(count: 160060), rate: tokens_per_second(count: 253), }], @@ -282,10 +281,11 @@ test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { // This is the arm that matters commercially, because the substitution is flattering in exactly one // direction: warm numbers are the fast ones, so letting a warm receipt answer a fresh question would // admit a realization that cannot serve a cold request at the floor. Here neither substitution is -// available, because the filter is regime equality and there is no widening arm to take. +// available. And the refusal is now STRUCTURAL rather than filtered: an observation carries no +// regime field at all, so a warm rate is not merely unmatched here, it is unwritable anywhere. The +// non-fresh arms of prefill_rate_at_floor have nothing to read. test fn w_a_fresh_prefill_receipt_does_not_answer_a_warm_continuation_floor() -> Bool { - let deep_fresh = [PrefillObservation { - regime: FreshSessionPrefill, + let deep_fresh = [FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135), }] @@ -353,8 +353,8 @@ data build_unmeasured: QuantizedCandidate = QuantizedCandidate { quality_rank: 4, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, - prefill_observations: [ - PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, + fresh_prefill_observations: [ + FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, ], kv_per_token: none, } @@ -398,8 +398,8 @@ data other_release: QuantizedCandidate = QuantizedCandidate { quality_rank: 3, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, - prefill_observations: [ - PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 900) }, + fresh_prefill_observations: [ + FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 900) }, ], kv_per_token: Present { value: byte_size(count: 20000) }, } @@ -464,8 +464,8 @@ data fixture_tied_a: QuantizedCandidate = QuantizedCandidate { quality_rank: 5, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, - prefill_observations: [ - PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + fresh_prefill_observations: [ + FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], kv_per_token: Present { value: byte_size(count: 20000) }, } @@ -477,8 +477,8 @@ data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { quality_rank: 5, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, - prefill_observations: [ - PrefillObservation { regime: FreshSessionPrefill, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + fresh_prefill_observations: [ + FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], kv_per_token: Present { value: byte_size(count: 20000) }, } From a70a8acea5eb2a6911b07662368d17a749053874 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 16:50:57 +0000 Subject: [PATCH 11/22] Memory fit is read off the running server, not multiplied out of an architecture constant The fixture declared kv_per_token = 88,064 bytes and called it "exact rather than estimated" because it was derived from the architecture -- 43 blocks, one latent KV head, key/value 512. It was a derivation wearing the label of a measurement, in the fixture for the module whose subject is refusing exactly that substitution. Loaded on spark-a3ee and read back through /api/ps size_vram, one build residents at 86,532,465,622 / 87,064,355,798 / 87,692,389,907 / 88,865,253,620 bytes for 131,072 / 262,144 / 400,000 / 1,048,576 tokens. Segment rates 4,058, 4,556, 1,808 bytes per token. THE FUNCTION IS FALSIFIED, NOT ITS COEFFICIENT, so no replacement scalar is authored: an averaged slope or a fitted curve would only manufacture a second unobserved answer with better provenance. The constant also fails alone -- at 88,064 B/token a 1,048,576 window is 92.3 GB of KV, exceeding the whole node beside 86.5 GB of weights, and that configuration loads and serves. So kv_per_token, kv_footprint, resident_footprint and the weights field are deleted, and fit is established from a ResidentFootprintObservation carrying realization, runtime, node, instrument, depth, concurrency and resident bytes. Deeper qualifies shallower and never the reverse, the tightest qualifying bound is returned, and CONCURRENCY IS NOT RECOVERED BY MULTIPLICATION -- a one-session reading cannot answer a two-session demand, so that is Unanswerable rather than scaled. The whole observation is returned rather than its byte count, so an established fit stays locatable. TWO RESULTS REVERSE. The higher-quality IQ3_S is not rejected on memory at a 400k floor; it residents in 116,970,000,000 B against 127,660,151,296 B allocatable. And fitting is not winning: it has no retrieval receipt and no prefill observation, so it is unresolved, and the real roster now REFUSES rather than handing the 2-bit build a default win. The evaluation-order annotation is corrected too. It claimed memory fit was decidable from declared facts, which is why ordering it first cost nothing; fit is now evidence-bearing and the memory arm can itself refuse, so the order buys refusal economy and not decidability. Also recorded: a fold accumulator bound by Present { value: x } does not carry its type to a field access, so the comparison is a named binary operation. That is a language-layer gap, noted where it bites rather than routed around silently. --- dag/gunbc/model/choice.dag | 134 ++++++++--- dag/gunbc/spark/serving_desired.dag | 28 ++- .../model/serving_choice_witness_test.dag | 216 +++++++++++++----- 3 files changed, 288 insertions(+), 90 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index 77252d080ba..639d3c6d49e 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -85,12 +85,11 @@ fn allocatable(capacity: NodeCapacity) -> ByteSize? { type QuantizedCandidate { identity: ReleaseIdentity quant_label: NonEmptyStr - weights: ByteSize quality_rank: Int declared_context: TokenCount semantic_context_verified_to: TokenCount? fresh_prefill_observations: List - kv_per_token: ByteSize? + resident_footprint_observations: List } // PREFILL RATE IS NOT A SCALAR, and this repository's own measurements falsified the scalar that @@ -178,16 +177,86 @@ fn slowest_fresh_rate_at_floor( }) } -// KV scales with context and with how many sessions are simultaneously RESIDENT. Sessions parked to -// disk between turns do not count -- their KV is not in memory -- which is why hot_sessions is the -// parameter rather than a total session count. -fn kv_footprint(per_token: ByteSize, context: TokenCount, hot_sessions: Nat) -> ByteSize { - byte_size(count: byte_size_count(b: per_token) * token_count_value(t: context) * hot_sessions) +// MEMORY FIT IS AN OBSERVATION, NOT A PRODUCT -- and the arithmetic that used to stand here is the +// reason this carrier exists. +// +// What stood here was `weights + kv_per_token * context * hot_sessions`, with kv_per_token fixed at +// 88,064 bytes and annotated "exact rather than estimated" because it was DERIVED from architecture: +// 43 blocks, one latent KV head, key and value lengths of 512. Read back off the running server, one +// build at four context depths resides in 86,532,465,622 / 87,064,355,798 / 87,692,389,907 / +// 88,865,253,620 bytes at 131,072 / 262,144 / 400,000 / 1,048,576 tokens. The segment rates are +// 4,058 then 4,556 then 1,808 bytes per token. +// +// THE FUNCTION IS FALSIFIED, NOT ITS COEFFICIENT. No scalar reproduces that curve, so replacing +// 88,064 with a better-sourced constant, an averaged slope or a fitted curve would only manufacture +// a second unobserved answer wearing better provenance. The derived figure also fails on its own +// terms: at 88,064 B/token a 1,048,576 window is 92.3 GB of KV, which with 86.5 GB of weights +// exceeds the node total -- and that configuration loads and serves. +// +// So a footprint is not computed. It is READ, at a stated depth, on a stated node, under a stated +// concurrency, by a stated instrument -- and each of those is part of the FACT rather than context +// around it, because a reading that does not carry them cannot be checked against the demand it is +// being used to answer. +type ResidentFootprintObservation { + realization: NonEmptyStr + runtime: NonEmptyStr + node: NonEmptyStr + instrument: NonEmptyStr + context_depth: TokenCount + concurrent_sessions: Nat + resident: ByteSize } -fn resident_footprint(weights: ByteSize, per_token: ByteSize, context: TokenCount, hot_sessions: Nat) -> ByteSize { - let kv = kv_footprint(per_token: per_token, context: context, hot_sessions: hot_sessions) - byte_size(count: byte_size_count(b: weights) + byte_size_count(b: kv)) +// The observation that answers a demand for `floor` context at `hot_sessions` resident sessions, or +// Absent when none does. +// +// DEEPER QUALIFIES SHALLOWER; SHALLOWER NEVER QUALIFIES DEEPER. That direction is the single +// assumption this function makes, and it is stated rather than buried: footprint is monotone in +// depth, so a reading at or beyond the floor upper-bounds the floor cost, while a shallower reading +// bounds nothing and flatters the candidate. Among qualifying readings the SMALLEST is the tightest +// sound bound, so that is the one returned. +// +// CONCURRENCY IS NOT RECOVERED BY MULTIPLICATION. A one-session reading says nothing executable +// about two simultaneously resident sessions -- shared weights, allocator behaviour and per-session +// caches do not decompose into a per-session term this instrument can see. A demand above the +// observed concurrency is UNANSWERABLE, exactly as a deeper floor is, and never a one-session number +// scaled up. +// +// THE WHOLE OBSERVATION IS RETURNED, not its byte count. Detaching `resident` would strip the depth, +// node, runtime and instrument that make the answer checkable, and a fit established from a reading +// nobody can locate again is not established. +fn resident_footprint_at_floor( + observations: List, + floor: TokenCount, + hot_sessions: Nat, +) -> ResidentFootprintObservation? { + let qualifying = filter(observations, o => + token_count_value(t: o.context_depth) >= token_count_value(t: floor) + && o.concurrent_sessions >= hot_sessions) + let tightest: ResidentFootprintObservation? = none + fold(qualifying, tightest, (best, o) => + match best { + Absent => Present { value: o } + Present { value: incumbent } => Present { value: tighter_bound(a: o, b: incumbent) } + }) +} + +// The comparison is a NAMED BINARY OPERATION rather than an inline match inside the fold, and the +// reason is a compiler limitation worth recording rather than routing silently around: a value bound +// by a `Present { value: x }` pattern over a fold accumulator does not carry its type to a FIELD +// ACCESS -- `x.resident` resolves as `no field 'resident' on type 'Unit'` -- while the same binding +// passed as a typed function ARGUMENT unifies correctly, which is why the prefill fold above never +// hit it. Seeding the fold with an annotated `none` does not repair the inference. So the accumulator +// type is established here by the parameter list. This is a real gap in accumulator type propagation, +// not a stylistic preference, and it will keep shaping folds until it is fixed at the language layer. +fn tighter_bound( + a: ResidentFootprintObservation, + b: ResidentFootprintObservation, +) -> ResidentFootprintObservation { + match byte_size_count(b: a.resident) < byte_size_count(b: b.resident) { + true => a + false => b + } } // ============================ CONSTRAINTS ============================ @@ -212,14 +281,18 @@ type ServingConstraints { // carry restated what the constructor already says, and a second malformation would need its own arm // rather than a different sentence. Prose whose content the type fixes is dead data (DESIGN 4c). type MissingFact - = KvFootprintUnmeasured + = ResidentFootprintUnmeasuredAtFloor { floor: TokenCount, sessions: Nat } | PrefillRateUnmeasuredAtFloor { floor: TokenCount, regime: ServingRegime } | SemanticContextUnverifiedAtFloor { floor: TokenCount, declared: TokenCount } | CapacityMalformed fn missing_fact_wire(missing: MissingFact) -> String { match missing { - KvFootprintUnmeasured => "kv-footprint-unmeasured" + ResidentFootprintUnmeasuredAtFloor { floor: f, sessions: n } => + join([ + "no resident-footprint reading at or beyond ", to_string(token_count_value(t: f)), + " tokens with at least ", to_string(n), " resident session(s)", + ], "") PrefillRateUnmeasuredAtFloor { floor: f, regime: g } => join([ "no ", serving_regime_wire(regime: g), " observation at or beyond ", @@ -251,11 +324,18 @@ type CandidateVerdict // Evaluated in a fixed order so the reported axis is deterministic: capability floors first, then // fit. A candidate that fails several axes reports the first, and the ordering puts the axis the // operator can act on soonest at the front. -// EVALUATION ORDER IS LOAD-BEARING, not stylistic. Memory fit is decidable from weights, KV cost and -// capacity alone, so it is settled FIRST: a candidate that cannot be resident does not need a -// retrieval receipt or a prefill observation, and demanding evidence before rejecting it would turn -// a decidable no into an unanswerable maybe. Only once a candidate could actually serve do we ask -// whether it is EVIDENCED to serve at this floor. +// EVALUATION ORDER IS LOAD-BEARING, not stylistic: memory fit is settled FIRST, because a candidate +// that cannot be resident does not need a retrieval receipt or a prefill observation to be refused. +// +// THE JUSTIFICATION FOR THAT ORDER CHANGED, and saying so is the point. It used to read that memory +// fit was DECIDABLE from weights, KV cost and capacity alone, so ordering it first turned no +// decidable answer into an unanswerable one. That is no longer true and was never true: fit is now +// established from an observation, so the memory arm can itself return Unanswerable, and the order +// buys refusal economy rather than decidability. A candidate with no qualifying footprint reading is +// unresolved on the FIRST axis, not admitted past it -- which is why the roster answer for a +// partially measured candidate is a refusal to select and not a default win for whoever was +// measured. Only once a candidate is evidenced to be resident do we ask whether it is evidenced to +// RETRIEVE and to SERVE at this floor. // // The context test consumes semantic_context_verified_to -- an observed retrieval depth -- and NEVER // declared_context. A publisher declaring 1,048,576 positions has made a claim about what the @@ -273,19 +353,21 @@ fn evaluate_candidate( quant_label: candidate.quant_label, missing: CapacityMalformed, } - Present { value: room } => match candidate.kv_per_token { + Present { value: room } => match resident_footprint_at_floor( + observations: candidate.resident_footprint_observations, + floor: constraints.context_floor, + hot_sessions: constraints.hot_sessions, + ) { Absent => ServingCandidateUnanswerable { identity: candidate.identity, quant_label: candidate.quant_label, - missing: KvFootprintUnmeasured, + missing: ResidentFootprintUnmeasuredAtFloor { + floor: constraints.context_floor, + sessions: constraints.hot_sessions, + }, } - Present { value: per_token } => { - let required = resident_footprint( - weights: candidate.weights, - per_token: per_token, - context: constraints.context_floor, - hot_sessions: constraints.hot_sessions, - ) + Present { value: observed } => { + let required = observed.resident match byte_size_count(b: required) > byte_size_count(b: room) { true => ServingCandidateRejected { identity: candidate.identity, diff --git a/dag/gunbc/spark/serving_desired.dag b/dag/gunbc/spark/serving_desired.dag index 6fb1cf1fa77..7ea0473de46 100644 --- a/dag/gunbc/spark/serving_desired.dag +++ b/dag/gunbc/spark/serving_desired.dag @@ -249,13 +249,27 @@ fn spark_serving_desired_launch_profile() -> OllamaServingLaunchProfile { // live window 16x on the first converge, as a side effect of a repair whose subject was causality // and not capacity. // -// 131072 is SELECTED here, and the wet observation is its EVIDENCE, not its source: both hosts were -// read reporting `context_length: 131072` from the Ollama /api/ps surface, which establishes that -// this value preserves what the running service already exposes. Observed state does not silently -// become desired state -- an operator chose this number and the reading is why it is defensible. A -// smaller window remains writable, and a larger one than the model declares is a real request the -// runtime is free to refuse. -data spark_serving_desired_context_length: TokenCount = token_count(count: 131072) +// THE 131072 THAT STOOD HERE WAS ALSO NEVER ENACTED, and its own evidence sentence is how that went +// unnoticed. It read that the wet observation was EVIDENCE and not source, citing both hosts +// reporting `context_length: 131072` from /api/ps -- but /api/ps reports what a LOADED MODEL +// ALLOCATED, which this file states two paragraphs above is a different fact from the CONFIGURED +// limit and must not be collapsed into it. An effective reading was therefore admitted as evidence +// for a configured value, and it could not have discriminated: the deployed unit on both hosts +// carried NO `OLLAMA_CONTEXT_LENGTH` line at all, so the number being confirmed was Ollama's own +// default and not this declaration. Measured 2026-09-01 by reading the unit file directly on +// spark-a3ee and spark-3bd5: three Environment lines, none of them the context one. +// +// 1048576 IS SELECTED, and what makes it defensible is a load rather than a reading: the served +// build hf.co/antirez/deepseek-v4-gguf declares `deepseek4.context_length` 1048576, and at that +// full window it resides in 88,865,253,620 B against a 130,660,151,296 B kernel-visible total. The +// window is not what is scarce on this hardware. Re-derive by loading the model at an explicit +// num_ctx and reading size_vram back from /api/ps; that instrument, not this number, is the +// authority for what any given build costs. +// +// A smaller window remains writable, and a larger one than the model declares is a real request +// the runtime is free to refuse -- Ollama clamps to the model's declared maximum, so this value is +// a CEILING across the roster and not a per-model promise. +data spark_serving_desired_context_length: TokenCount = token_count(count: 1048576) // The unit's window is DERIVED from the serving policy rather than authored beside it. This is the // causal link the repair adds: the policy is the single authority, the unit text is a function of diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index acc6e256a95..66c28b2f907 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -11,15 +11,16 @@ import gunbc.model.publication { ReleaseIdentity } import gunbc.model.choice { PrefillRateUnmeasuredAtFloor, SemanticContextUnverifiedAtFloor, FreshPrefillObservation, prefill_rate_at_floor, + ResidentFootprintObservation, resident_footprint_at_floor, FreshSessionPrefill, WarmContinuation, serving_regime_wire, NodeCapacity, QuantizedCandidate, ServingConstraints, CandidateVerdict, ServingCandidateAdmissible, ServingCandidateRejected, RejectionAxis, DoesNotFitMemory, ContextBelowFloor, PrefillBelowFloor, ServingChoice, ChoseCandidate, NoCandidateAdmissible, SelectionUnanswerable, UnanswerableCause, UnresolvedCandidateCouldWin, CrossReleaseQualityOrderAbsent, QualityRankTie, - ServingCandidateUnanswerable, MissingFact, KvFootprintUnmeasured, CapacityMalformed, + ServingCandidateUnanswerable, MissingFact, ResidentFootprintUnmeasuredAtFloor, CapacityMalformed, evaluate_candidate, choose_serving_candidate, - allocatable, firmware_carveout, kv_footprint, + allocatable, firmware_carveout, } // THE FIXTURE IS MEASURED, not invented. Every number here was read off the running nodes or @@ -30,8 +31,16 @@ import gunbc.model.choice { // nominal 137,438,953,472 B = 128 GiB (vendor sheet) // carveout 6,778,802,176 B = 6.31 GiB (firmware, never reaches the kernel) // -// KV is exact rather than estimated: the architecture reports 43 blocks, ONE latent KV head (MLA), -// and key/value lengths of 512, so a token costs 43 * 1 * (512+512) * 2 = 88,064 bytes. +// KV WAS NOT EXACT, AND THIS BLOCK IS WHY THE MODULE NOW REFUSES TO COMPUTE FOOTPRINTS AT ALL. +// It read: "KV is exact rather than estimated -- 43 blocks, ONE latent KV head (MLA), key/value +// lengths of 512, so a token costs 43 * 1 * (512+512) * 2 = 88,064 bytes." That number is DERIVED +// from architecture and was labelled measured, in the fixture for a module whose subject is refusing +// exactly that substitution. Loading the build and reading it back falsified both the constant and +// the linear form it was multiplied through (see gunbc.model.choice ResidentFootprintObservation). +// +// Every footprint below is now a reading. Re-derive any of them by loading the build at an explicit +// num_ctx and reading `size_vram` back from the Ollama /api/ps surface on the named node; that +// instrument, not any number transcribed here, is the authority. data node_capacity: NodeCapacity = NodeCapacity { node_name: "spark-a3ee" as NonEmptyStr, nominal: byte_size(count: 137438953472), @@ -39,7 +48,19 @@ data node_capacity: NodeCapacity = NodeCapacity { runtime_overhead: byte_size(count: 3000000000), } -data kv_per_token: ByteSize = byte_size(count: 88064) +// THE READINGS, spark-a3ee, Ollama 0.32.9, one resident session, /api/ps size_vram. `size` and +// `size_vram` agreed exactly on this unified-memory node, so nothing turns on which field is read. +fn iq2_xxs_footprint(depth: Nat, resident: Nat) -> ResidentFootprintObservation { + ResidentFootprintObservation { + realization: "hf.co/antirez/deepseek-v4-gguf:latest" as NonEmptyStr, + runtime: "ollama 0.32.9" as NonEmptyStr, + node: "spark-a3ee" as NonEmptyStr, + instrument: "/api/ps size_vram after load at explicit num_ctx" as NonEmptyStr, + context_depth: token_count(count: depth), + concurrent_sessions: 1, + resident: byte_size(count: resident), + } +} // RELEASE IDENTITIES, exact rather than display names. Publisher, family and revision together are // what decide whether two candidates share an ordinal quality scale, so the selector is keyed by @@ -69,7 +90,6 @@ data qwen_release: ReleaseIdentity = ReleaseIdentity { data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { identity: deepseek_v4_flash_release, quant_label: "IQ2_XXS" as NonEmptyStr, - weights: byte_size(count: 86720111200), quality_rank: 2, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, @@ -78,18 +98,32 @@ data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { FreshPrefillObservation { depth: token_count(count: 255061), rate: tokens_per_second(count: 197) }, FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, ], - kv_per_token: Present { value: kv_per_token }, + resident_footprint_observations: [ + iq2_xxs_footprint(depth: 131072, resident: 86532465622), + iq2_xxs_footprint(depth: 262144, resident: 87064355798), + iq2_xxs_footprint(depth: 400000, resident: 87692389907), + iq2_xxs_footprint(depth: 1048576, resident: 88865253620), + ], } data build_iq3_s: QuantizedCandidate = QuantizedCandidate { identity: deepseek_v4_flash_release, quant_label: "UD-IQ3_S" as NonEmptyStr, - weights: byte_size(count: 116100000000), quality_rank: 3, declared_context: token_count(count: 1048576), semantic_context_verified_to: none, fresh_prefill_observations: [], - kv_per_token: Present { value: kv_per_token }, + resident_footprint_observations: [ + ResidentFootprintObservation { + realization: "deepseek-v4-flash:iq3s" as NonEmptyStr, + runtime: "ollama 0.32.9" as NonEmptyStr, + node: "spark-a3ee" as NonEmptyStr, + instrument: "/api/ps size_vram after load at explicit num_ctx" as NonEmptyStr, + context_depth: token_count(count: 400000), + concurrent_sessions: 1, + resident: byte_size(count: 116970000000), + }, + ], } fn installed() -> List { @@ -143,32 +177,77 @@ test fn w_capacity_grains_differ_and_only_kernel_visible_is_spendable() -> Bool } } -// KV at the operator's floor, from the model's own dimensions: 400,000 * 88,064 = 35.2 GB. -test fn w_kv_at_the_context_floor_is_derived_from_architecture() -> Bool { - byte_size_count(b: kv_footprint( - per_token: kv_per_token, context: token_count(count: 400000), hot_sessions: 1)) == 35225600000 -} - -// THE CLAIM THE MODULE EXISTS FOR. At a 400k floor the HIGHER-quality build is REJECTED ON MEMORY -// and the selector returns the 2-bit one. This is the answer that reversed the session's working -// assumption -- precision and the context floor compete for one node's memory, and the selector -// says so from the numbers rather than from an argument. -test fn w_at_the_400k_floor_the_higher_quality_build_is_rejected_on_memory() -> Bool { +// THE CLAIM THE MODULE EXISTS FOR, AND THE ONE IT GOT WRONG. +// +// What stood here asserted that at a 400k floor the higher-quality IQ3_S is REJECTED ON MEMORY and +// the 2-bit build wins -- reported as the result that reversed the session's working assumption, +// that precision and context compete for one node's memory. It was an artifact of the falsified +// 88,064 B/token constant, which charged 35.2 GB of KV at that floor. Loaded on the node, IQ3_S +// residents a 400,000 window in 116,970,000,000 B against 127,660,151,296 B allocatable. IT FITS. +// +// AND FITTING IS NOT WINNING, which is the second half of the correction and the more expensive one +// to have missed. Reversing a memory rejection does not promote the candidate; it moves it off the +// first axis and onto the next, where it has no evidence at all. IQ3_S carries no retrieval receipt +// and no prefill observation, so it is UNANSWERABLE on semantic context. Unknown is not qualified, +// exactly as unknown is not rejected. +test fn w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolved() -> Bool { match evaluate_candidate(candidate: build_iq3_s, capacity: node_capacity, constraints: floor_400k) { - ServingCandidateRejected { identity: _, quant_label: _, axis: a } => - match a { - DoesNotFitMemory { required: _, allocatable: _ } => true - ContextBelowFloor { declared: _, floor: _ } => false - PrefillBelowFloor { measured: _, floor: _ } => false + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => + match m { + SemanticContextUnverifiedAtFloor { floor: f, declared: d } => + token_count_value(t: f) == 400000 && token_count_value(t: d) == 1048576 + ResidentFootprintUnmeasuredAtFloor { floor: _, sessions: _ } => false + PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false + CapacityMalformed => false } + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false - ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => false } } -test fn w_at_the_400k_floor_the_two_bit_build_is_chosen() -> Bool { - chosen_label(choice: choose_serving_candidate( - candidates: installed(), capacity: node_capacity, constraints: floor_400k)) == "IQ2_XXS" +// SO THE REAL ROSTER HAS NO WINNER. IQ2_XXS is measured on every axis and IQ3_S outranks it while +// being unresolved, so a selection would be a default win for whoever happened to be measured. The +// selector refuses, and the refusal names the unresolved candidate as the reason. +test fn w_the_real_roster_refuses_while_the_higher_rank_is_unresolved() -> Bool { + match choose_serving_candidate( + candidates: installed(), capacity: node_capacity, constraints: floor_400k) { + SelectionUnanswerable { cause: _, evaluations: e } => length(e) == 2 + ChoseCandidate { verdict: _ } => false + NoCandidateAdmissible { rejections: _ } => false + } +} + +// THE POSITIVE CONTROL FOR THE FOOTPRINT AXIS: the 2-bit build IS resident-established at the floor, +// so the refusal above is not a footprint carrier that answers nothing. Its qualifying reading is +// the 400,000 one rather than the 1,048,576 one -- the tightest sound bound at or beyond the floor. +test fn w_the_two_bit_build_is_resident_established_at_the_floor() -> Bool { + match resident_footprint_at_floor( + observations: build_iq2_xxs.resident_footprint_observations, + floor: token_count(count: 400000), + hot_sessions: 1, + ) { + Present { value: o } => + byte_size_count(b: o.resident) == 87692389907 + && token_count_value(t: o.context_depth) == 400000 + Absent => false + } +} + +// A SHALLOWER READING DOES NOT QUALIFY A DEEPER FLOOR, and a demand above the observed concurrency +// is unanswerable rather than multiplied. Both arms discriminate against the same observation list +// that positively answers one hot session at 400k above. +test fn w_a_shallow_or_more_concurrent_demand_has_no_qualifying_footprint() -> Bool { + let obs = build_iq2_xxs.resident_footprint_observations + match resident_footprint_at_floor( + observations: obs, floor: token_count(count: 2000000), hot_sessions: 1) { + Present { value: _ } => false + Absent => + match resident_footprint_at_floor( + observations: obs, floor: token_count(count: 400000), hot_sessions: 2) { + Present { value: _ } => false + Absent => true + } + } } // THE DISCRIMINATING CONTROL, on a CONSTRUCTED pair rather than the installed builds. @@ -179,33 +258,46 @@ test fn w_at_the_400k_floor_the_two_bit_build_is_chosen() -> Bool { // was whether the runtime ACCEPTS a num_ctx value, which a two-token request answers without ever // allocating the cache, so it established nothing about retrieval. // +// SYNTHETIC footprints for the declared ordering fixtures below. These are not fleet readings and +// say so in every field: the node is named "fixture", so a real constraint can never be answered by +// one, and the ordering claims they support are about the SELECTOR and not about any hardware. +fn fixture_footprint(resident: Nat) -> ResidentFootprintObservation { + ResidentFootprintObservation { + realization: "fixture" as NonEmptyStr, + runtime: "fixture" as NonEmptyStr, + node: "fixture" as NonEmptyStr, + instrument: "declared fixture, not a reading" as NonEmptyStr, + context_depth: token_count(count: 1048576), + concurrent_sessions: 1, + resident: byte_size(count: resident), + } +} + // These two are declared fixtures exercising the ordering rule, NOT fleet observations. They share a // release, so a maximum over them exists, and they differ only in quality_rank and weights. Without // this control every other claim here is satisfied by a function that returns its first argument. data fixture_low_rank: QuantizedCandidate = QuantizedCandidate { identity: fixture_release, quant_label: "LOW" as NonEmptyStr, - weights: byte_size(count: 20000000000), quality_rank: 1, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, fresh_prefill_observations: [ FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], - kv_per_token: Present { value: byte_size(count: 20000) }, + resident_footprint_observations: [fixture_footprint(resident: 28000000000)], } data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { identity: fixture_release, quant_label: "HIGH" as NonEmptyStr, - weights: byte_size(count: 30000000000), quality_rank: 9, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, fresh_prefill_observations: [ FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 400) }, ], - kv_per_token: Present { value: byte_size(count: 20000) }, + resident_footprint_observations: [fixture_footprint(resident: 38000000000)], } test fn w_the_higher_quality_candidate_wins_when_both_are_admissible() -> Bool { @@ -227,8 +319,8 @@ test fn w_a_declared_context_never_qualifies_without_a_retrieval_receipt() -> Bo match m { SemanticContextUnverifiedAtFloor { floor: _, declared: d } => token_count_value(t: d) == 1048576 - KvFootprintUnmeasured => false - PrefillRateUnmeasuredAtFloor { floor: _ } => false + ResidentFootprintUnmeasuredAtFloor { floor: _, sessions: _ } => false + PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false CapacityMalformed => false } ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false @@ -255,15 +347,27 @@ test fn w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() -> Bool // When nothing is admissible the selector REFUSES and hands back every rejection, so the binding // axis is read rather than argued. A best-effort pick here would reintroduce the silent degradation // the whole exercise removes. +// +// IT IS RUN ON A STARVED NODE, NOT ON AN ABSURD FLOOR, and that change is a consequence of the +// footprint repair rather than a convenience. This claim used to ask for an 8,000,000-token floor, +// which under the old arithmetic multiplied out to a memory REJECTION. There is no footprint reading +// at that depth and never will be one, so the honest answer there is now UNANSWERABLE -- the +// candidates have not failed the test, they have not taken it. To witness rejection the candidates +// must be evidenced and the room must be genuinely too small, so the capacity shrinks instead of the +// floor growing. The fixtures are used because they are resident-established at depth by +// construction; the installed builds are not, at any depth beyond what was actually loaded. test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { - let impossible = ServingConstraints { - context_floor: token_count(count: 8000000), - hot_sessions: 1, - prefill_floor: tokens_per_second(count: 0), - prefill_regime: FreshSessionPrefill, + let starved = NodeCapacity { + node_name: "fixture-starved" as NonEmptyStr, + nominal: byte_size(count: 34000000000), + kernel_visible: byte_size(count: 30000000000), + runtime_overhead: byte_size(count: 3000000000), } match choose_serving_candidate( - candidates: installed(), capacity: node_capacity, constraints: impossible) { + candidates: [fixture_low_rank, fixture_high_rank], + capacity: starved, + constraints: floor_400k, + ) { NoCandidateAdmissible { rejections: r } => length(r) == 2 ChoseCandidate { verdict: _ } => false SelectionUnanswerable { cause: _, evaluations: _ } => false @@ -317,7 +421,7 @@ test fn w_a_warm_floor_over_fresh_only_receipts_is_unanswerable() -> Bool { match m { PrefillRateUnmeasuredAtFloor { floor: _, regime: g } => serving_regime_wire(regime: g) == "warm-continuation" - KvFootprintUnmeasured => false + ResidentFootprintUnmeasuredAtFloor { floor: _, sessions: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false CapacityMalformed => false } @@ -328,9 +432,10 @@ test fn w_a_warm_floor_over_fresh_only_receipts_is_unanswerable() -> Bool { test fn w_all_serving_choice_claims_hold() -> Bool { w_capacity_grains_differ_and_only_kernel_visible_is_spendable() - && w_kv_at_the_context_floor_is_derived_from_architecture() - && w_at_the_400k_floor_the_higher_quality_build_is_rejected_on_memory() - && w_at_the_400k_floor_the_two_bit_build_is_chosen() + && w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolved() + && w_the_real_roster_refuses_while_the_higher_rank_is_unresolved() + && w_the_two_bit_build_is_resident_established_at_the_floor() + && w_a_shallow_or_more_concurrent_demand_has_no_qualifying_footprint() && w_the_higher_quality_candidate_wins_when_both_are_admissible() && w_a_declared_context_never_qualifies_without_a_retrieval_receipt() && w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() @@ -349,22 +454,22 @@ test fn w_all_serving_choice_claims_hold() -> Bool { data build_unmeasured: QuantizedCandidate = QuantizedCandidate { identity: deepseek_v4_flash_release, quant_label: "UD-IQ4_XS" as NonEmptyStr, - weights: byte_size(count: 136700000000), quality_rank: 4, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, fresh_prefill_observations: [ FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, ], - kv_per_token: none, + resident_footprint_observations: [], } test fn w_an_unmeasured_candidate_is_unanswerable_not_rejected() -> Bool { match evaluate_candidate(candidate: build_unmeasured, capacity: node_capacity, constraints: floor_400k) { ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => match m { - KvFootprintUnmeasured => true - PrefillRateUnmeasuredAtFloor { floor: _ } => false + ResidentFootprintUnmeasuredAtFloor { floor: f, sessions: n } => + token_count_value(t: f) == 400000 && n == 1 + PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false CapacityMalformed => false } @@ -394,14 +499,13 @@ test fn w_selection_refuses_while_an_unresolved_candidate_could_win() -> Bool { data other_release: QuantizedCandidate = QuantizedCandidate { identity: qwen_release, quant_label: "Q8_0" as NonEmptyStr, - weights: byte_size(count: 22600000000), quality_rank: 3, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, fresh_prefill_observations: [ FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 900) }, ], - kv_per_token: Present { value: byte_size(count: 20000) }, + resident_footprint_observations: [fixture_footprint(resident: 30600000000)], } test fn w_two_admissible_releases_refuse_for_want_of_a_cross_release_order() -> Bool { @@ -436,8 +540,8 @@ test fn w_malformed_capacity_is_unanswerable_not_a_memory_rejection() -> Bool { ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => match m { CapacityMalformed => true - KvFootprintUnmeasured => false - PrefillRateUnmeasuredAtFloor { floor: _ } => false + ResidentFootprintUnmeasuredAtFloor { floor: _, sessions: _ } => false + PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false } ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false @@ -460,27 +564,25 @@ test fn w_all_answerability_claims_hold() -> Bool { data fixture_tied_a: QuantizedCandidate = QuantizedCandidate { identity: fixture_release, quant_label: "TIED-A" as NonEmptyStr, - weights: byte_size(count: 20000000000), quality_rank: 5, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, fresh_prefill_observations: [ FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], - kv_per_token: Present { value: byte_size(count: 20000) }, + resident_footprint_observations: [fixture_footprint(resident: 28000000000)], } data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { identity: fixture_release, quant_label: "TIED-B" as NonEmptyStr, - weights: byte_size(count: 21000000000), quality_rank: 5, declared_context: token_count(count: 1048576), semantic_context_verified_to: Present { value: token_count(count: 400060) }, fresh_prefill_observations: [ FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], - kv_per_token: Present { value: byte_size(count: 20000) }, + resident_footprint_observations: [fixture_footprint(resident: 29000000000)], } fn is_tie_refusal(choice: ServingChoice) -> Bool { From 8917397e6278b3e50f923ef872e9a017f59fa14c Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 18:11:41 +0000 Subject: [PATCH 12/22] Move the serving context bump out of the model-authority PR The 1,048,576 desired-context change is a FLEET change: it alters what the two Sparks are configured to serve. This PR's subject is the model selection authority. They were riding together only because I made both edits in the same hour, which is not a reason. It lands on its own branch with the rung drop it needs -- the measurement behind 1M is bound to one node, one IQ2_XXS artifact, one runtime and one concurrency condition, and it is being applied to two hosts and every realization on the roster. That declaration does not belong in a diff about ReleaseIdentity. --- dag/gunbc/spark/serving_desired.dag | 28 +++++++--------------------- 1 file changed, 7 insertions(+), 21 deletions(-) diff --git a/dag/gunbc/spark/serving_desired.dag b/dag/gunbc/spark/serving_desired.dag index 7ea0473de46..6fb1cf1fa77 100644 --- a/dag/gunbc/spark/serving_desired.dag +++ b/dag/gunbc/spark/serving_desired.dag @@ -249,27 +249,13 @@ fn spark_serving_desired_launch_profile() -> OllamaServingLaunchProfile { // live window 16x on the first converge, as a side effect of a repair whose subject was causality // and not capacity. // -// THE 131072 THAT STOOD HERE WAS ALSO NEVER ENACTED, and its own evidence sentence is how that went -// unnoticed. It read that the wet observation was EVIDENCE and not source, citing both hosts -// reporting `context_length: 131072` from /api/ps -- but /api/ps reports what a LOADED MODEL -// ALLOCATED, which this file states two paragraphs above is a different fact from the CONFIGURED -// limit and must not be collapsed into it. An effective reading was therefore admitted as evidence -// for a configured value, and it could not have discriminated: the deployed unit on both hosts -// carried NO `OLLAMA_CONTEXT_LENGTH` line at all, so the number being confirmed was Ollama's own -// default and not this declaration. Measured 2026-09-01 by reading the unit file directly on -// spark-a3ee and spark-3bd5: three Environment lines, none of them the context one. -// -// 1048576 IS SELECTED, and what makes it defensible is a load rather than a reading: the served -// build hf.co/antirez/deepseek-v4-gguf declares `deepseek4.context_length` 1048576, and at that -// full window it resides in 88,865,253,620 B against a 130,660,151,296 B kernel-visible total. The -// window is not what is scarce on this hardware. Re-derive by loading the model at an explicit -// num_ctx and reading size_vram back from /api/ps; that instrument, not this number, is the -// authority for what any given build costs. -// -// A smaller window remains writable, and a larger one than the model declares is a real request -// the runtime is free to refuse -- Ollama clamps to the model's declared maximum, so this value is -// a CEILING across the roster and not a per-model promise. -data spark_serving_desired_context_length: TokenCount = token_count(count: 1048576) +// 131072 is SELECTED here, and the wet observation is its EVIDENCE, not its source: both hosts were +// read reporting `context_length: 131072` from the Ollama /api/ps surface, which establishes that +// this value preserves what the running service already exposes. Observed state does not silently +// become desired state -- an operator chose this number and the reading is why it is defensible. A +// smaller window remains writable, and a larger one than the model declares is a real request the +// runtime is free to refuse. +data spark_serving_desired_context_length: TokenCount = token_count(count: 131072) // The unit's window is DERIVED from the serving policy rather than authored beside it. This is the // causal link the repair adds: the policy is the single authority, the unit text is a function of From c5cb567528db203645d70243c64f568ae67f261a Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 18:35:40 +0000 Subject: [PATCH 13/22] A runner buffer figure is not an OS resident set, so the selector stops subtracting one from the other The repair I pushed for the falsified KV constant introduced a subtler version of the same error, and review caught it before it decided anything. The carrier was named ResidentFootprintObservation with a field `resident`, and that number was compared against allocatable(capacity) -- kernel-visible memory from /proc/meminfo less a declared overhead. /api/ps reports `size` and `size_vram`, which Ollama documents as total and GPU BUFFER usage parsed from its runner's logs. That is not an operating-system resident set. The comparison spanned two accounting domains, so the headroom it produced was manufactured, and the name asserted the bridge that no measurement had crossed -- the same move as labelling a derived constant "exact", one layer up. So fit is no longer computed and no longer compared. IT IS EXECUTED. The carrier is OllamaRunnerMemoryObservation, capturing both reported buffer figures without renaming either into an OS quantity, plus the realization, runtime, runtime configuration, node, instrument, depth, concurrency, and whether generation SUCCEEDED. A successful load is a positive fit under exactly the conditions that held; a failed one is an observed non-fit; no observation is Unanswerable. Node equality joins the qualification, because a fit established on one machine says nothing about another. DELETED WITH THE COMPARISON: NodeCapacity, allocatable, firmware_carveout, CapacityMalformed, and the two witnesses whose subject they were. The three-grain distinction between vendor nominal, kernel-visible and live-available is real and has already cost one wrong decision, but it belongs to an authority that owns node facts, not to a selector that can no longer compare against them. Keeping the carrier so its witness had something to test would be a check keeping its own subject alive. The starved-node rejection witness moved for the second time. It first used an absurd context floor, which became Unanswerable once fit needed evidence; then a starved NodeCapacity, which is gone with the subtraction. What remains is the only non-fit the instrument can report: a configuration attempted on the node that did not produce a token. Every witness aggregate re-run and passing: serving choice, answerability, tie refusal, population. --- dag/gunbc/model/choice.dag | 198 +++++++------- .../model/serving_choice_witness_test.dag | 258 ++++++++++-------- 2 files changed, 238 insertions(+), 218 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index 639d3c6d49e..682bc1ede2a 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -37,44 +37,21 @@ import std.measure { // ============================ CAPACITY, at the three grains that actually differ ============================ -// A vendor's nominal figure, the kernel's visible total, and a live availability reading are THREE -// DIFFERENT QUANTITIES, and confusing them has already produced a wrong fit decision. Nominal -// overstates by the firmware carveout; observed-available understates by whatever is resident at the -// moment of the reading. Only kernel_visible is a capacity, so only it is consumed here -- -// the others are retained so a reader can see the gap rather than rediscover it. -type NodeCapacity { - node_name: NonEmptyStr - nominal: ByteSize - kernel_visible: ByteSize - runtime_overhead: ByteSize -} - -// SUBTRACTION OVER Nat TRAPS, so both of these answer with an option rather than assuming their -// operands are ordered. A vendor sheet smaller than the kernel's own reading, or an overhead -// allowance exceeding the machine, is a MALFORMED CAPACITY -- a fact about the inputs, not a verdict -// about any candidate -- and the selector reports it as unanswerable rather than trapping or -// silently clamping to zero. Clamping would be the absorbing-fallback shape: an impossible machine -// would quietly become a machine with no room, and every candidate would be rejected on memory for -// a reason that was never true. -fn firmware_carveout(capacity: NodeCapacity) -> ByteSize? { - match byte_size_count(b: capacity.kernel_visible) > byte_size_count(b: capacity.nominal) { - true => none - false => Present { - value: byte_size(count: byte_size_count(b: capacity.nominal) - byte_size_count(b: capacity.kernel_visible)), - } - } -} - -// The bytes a model may actually occupy: what the kernel can hand out, less what the serving runtime -// and OS need to stay alive. Deliberately NOT derived from a live availability sample. -fn allocatable(capacity: NodeCapacity) -> ByteSize? { - match byte_size_count(b: capacity.runtime_overhead) > byte_size_count(b: capacity.kernel_visible) { - true => none - false => Present { - value: byte_size(count: byte_size_count(b: capacity.kernel_visible) - byte_size_count(b: capacity.runtime_overhead)), - } - } -} +// THE NODE CAPACITY CARRIER USED TO STAND HERE, with a three-grain distinction between a vendor's +// nominal figure, the kernel's visible total and a live availability reading, and an `allocatable` +// derivation that subtracted a declared runtime overhead from the kernel total. +// +// IT IS DELETED BECAUSE NOTHING IN THIS MODULE CAN CONSUME IT ANY MORE. The only decision it fed was +// a comparison against an observed footprint, and that comparison turned out to span two accounting +// domains: the observed quantity is an Ollama runner BUFFER figure parsed from llama-server logs, +// and this was a /proc/meminfo total. Their difference is not a fact, so the headroom it produced was +// manufactured. Once the memory arm stopped subtracting, capacity had no consumer, and a value the +// decision cannot use is decoration that a later reader will mistake for a constraint. +// +// The three-grain distinction is real and worth keeping -- it has already caused one wrong fit +// decision by treating a vendor nominal as spendable -- but it belongs to a capacity authority that +// owns node facts, not to a selector that can no longer compare against them. It is removed here +// rather than reproduced there speculatively, because this module is not that authority. // ============================ CANDIDATES ============================ @@ -89,7 +66,7 @@ type QuantizedCandidate { declared_context: TokenCount semantic_context_verified_to: TokenCount? fresh_prefill_observations: List - resident_footprint_observations: List + runtime_memory_observations: List } // PREFILL RATE IS NOT A SCALAR, and this repository's own measurements falsified the scalar that @@ -197,18 +174,37 @@ fn slowest_fresh_rate_at_floor( // concurrency, by a stated instrument -- and each of those is part of the FACT rather than context // around it, because a reading that does not carry them cannot be checked against the demand it is // being used to answer. -type ResidentFootprintObservation { +// THE CARRIER NAMES WHAT THE INSTRUMENT REPORTS AND NOTHING MORE. An earlier version of this type +// was called ResidentFootprintObservation with a field named `resident`, and that name asserted a +// bridge no measurement had crossed: /api/ps reports `size` and `size_vram`, which Ollama documents +// as total and GPU BUFFER USAGE parsed from its runner's logs -- not an operating-system resident +// set. Comparing that figure against a /proc/meminfo total minus a declared overhead spans two +// accounting domains, and the headroom such a subtraction produces is manufactured. Naming the +// field `resident` asserted the bridge in the NAME, which is the same move as calling a derived +// constant "exact"; it was caught in review before it decided anything. +// +// So both reported quantities are captured, neither is renamed into an OS quantity, and the fit +// claim rests on something the instrument CAN establish: whether the configuration loaded and +// produced a token on that node. On the measured unified-memory nodes the two figures were +// byte-identical in every reading, but that equality is carried here as data to be observed rather +// than assumed, because a full-offload receipt is what would license treating them as one. +type OllamaRunnerMemoryObservation { realization: NonEmptyStr runtime: NonEmptyStr + runtime_config: NonEmptyStr node: NonEmptyStr instrument: NonEmptyStr context_depth: TokenCount concurrent_sessions: Nat - resident: ByteSize + reported_total_buffer: ByteSize + reported_gpu_buffer: ByteSize + generation_succeeded: Bool } -// The observation that answers a demand for `floor` context at `hot_sessions` resident sessions, or -// Absent when none does. +// The observation that answers a demand for `floor` context at `hot_sessions` resident sessions ON +// THIS NODE, or Absent when none does. Node equality is a qualification and not a label: a fit +// established on one machine says nothing about another, and importing it would be the same +// substitution as importing a shallower depth. // // DEEPER QUALIFIES SHALLOWER; SHALLOWER NEVER QUALIFIES DEEPER. That direction is the single // assumption this function makes, and it is stated rather than buried: footprint is monotone in @@ -222,18 +218,24 @@ type ResidentFootprintObservation { // observed concurrency is UNANSWERABLE, exactly as a deeper floor is, and never a one-session number // scaled up. // -// THE WHOLE OBSERVATION IS RETURNED, not its byte count. Detaching `resident` would strip the depth, +// THE WHOLE OBSERVATION IS RETURNED, not a byte count. Detaching a number would strip the depth, // node, runtime and instrument that make the answer checkable, and a fit established from a reading // nobody can locate again is not established. -fn resident_footprint_at_floor( - observations: List, +// +// SELECTION AMONG QUALIFYING OBSERVATIONS IS BY REPORTED BUFFER AND NOT BY ROSTER ORDER. The +// comparison is between two readings from ONE instrument, so it stays inside a single accounting +// domain -- which is exactly the property the discarded capacity comparison did not have. +fn runtime_memory_observation_at_floor( + observations: List, floor: TokenCount, hot_sessions: Nat, -) -> ResidentFootprintObservation? { + node: NonEmptyStr, +) -> OllamaRunnerMemoryObservation? { let qualifying = filter(observations, o => token_count_value(t: o.context_depth) >= token_count_value(t: floor) - && o.concurrent_sessions >= hot_sessions) - let tightest: ResidentFootprintObservation? = none + && o.concurrent_sessions >= hot_sessions + && (o.node as String) == (node as String)) + let tightest: OllamaRunnerMemoryObservation? = none fold(qualifying, tightest, (best, o) => match best { Absent => Present { value: o } @@ -250,10 +252,10 @@ fn resident_footprint_at_floor( // type is established here by the parameter list. This is a real gap in accumulator type propagation, // not a stylistic preference, and it will keep shaping folds until it is fixed at the language layer. fn tighter_bound( - a: ResidentFootprintObservation, - b: ResidentFootprintObservation, -) -> ResidentFootprintObservation { - match byte_size_count(b: a.resident) < byte_size_count(b: b.resident) { + a: OllamaRunnerMemoryObservation, + b: OllamaRunnerMemoryObservation, +) -> OllamaRunnerMemoryObservation { + match byte_size_count(b: a.reported_total_buffer) < byte_size_count(b: b.reported_total_buffer) { true => a false => b } @@ -276,22 +278,17 @@ type ServingConstraints { // A candidate whose KV cost was never measured has not failed the memory test -- it has not TAKEN // it. Reporting that as a rejection would let a fully-measured but worse candidate win by default, // and the answer would look identical to one where the better candidate genuinely lost. -// CapacityMalformed carries NO payload. There is exactly one way a NodeCapacity is malformed -- -// declared runtime overhead exceeding the kernel-visible total -- so the `detail: String` it used to -// carry restated what the constructor already says, and a second malformation would need its own arm -// rather than a different sentence. Prose whose content the type fixes is dead data (DESIGN 4c). type MissingFact - = ResidentFootprintUnmeasuredAtFloor { floor: TokenCount, sessions: Nat } + = MemoryFitUnobservedAtConfiguration { floor: TokenCount, sessions: Nat } | PrefillRateUnmeasuredAtFloor { floor: TokenCount, regime: ServingRegime } | SemanticContextUnverifiedAtFloor { floor: TokenCount, declared: TokenCount } - | CapacityMalformed fn missing_fact_wire(missing: MissingFact) -> String { match missing { - ResidentFootprintUnmeasuredAtFloor { floor: f, sessions: n } => + MemoryFitUnobservedAtConfiguration { floor: f, sessions: n } => join([ - "no resident-footprint reading at or beyond ", to_string(token_count_value(t: f)), - " tokens with at least ", to_string(n), " resident session(s)", + "no runner-memory observation on this node at or beyond ", to_string(token_count_value(t: f)), + " tokens with at least ", to_string(n), " concurrent session(s)", ], "") PrefillRateUnmeasuredAtFloor { floor: f, regime: g } => join([ @@ -304,7 +301,6 @@ fn missing_fact_wire(missing: MissingFact) -> String { "; the publisher declares ", to_string(token_count_value(t: d)), " but a declaration is a claim, not evidence", ], "") - CapacityMalformed => "capacity-malformed: declared runtime overhead exceeds the kernel-visible total" } } @@ -312,12 +308,12 @@ fn missing_fact_wire(missing: MissingFact) -> String { // census over rejections says which INPUT would have to move -- usable memory, a declared floor, a // published artifact -- without claiming which purchase moves it. type RejectionAxis - = DoesNotFitMemory { required: ByteSize, allocatable: ByteSize } + = DoesNotFitMemory { observed: OllamaRunnerMemoryObservation } | ContextBelowFloor { declared: TokenCount, floor: TokenCount } | PrefillBelowFloor { measured: TokensPerSecond, floor: TokensPerSecond } type CandidateVerdict - = ServingCandidateAdmissible { candidate: QuantizedCandidate, resident: ByteSize, headroom: ByteSize } + = ServingCandidateAdmissible { candidate: QuantizedCandidate, fit: OllamaRunnerMemoryObservation } | ServingCandidateRejected { identity: ReleaseIdentity, quant_label: NonEmptyStr, axis: RejectionAxis } | ServingCandidateUnanswerable { identity: ReleaseIdentity, quant_label: NonEmptyStr, missing: MissingFact } @@ -327,15 +323,29 @@ type CandidateVerdict // EVALUATION ORDER IS LOAD-BEARING, not stylistic: memory fit is settled FIRST, because a candidate // that cannot be resident does not need a retrieval receipt or a prefill observation to be refused. // -// THE JUSTIFICATION FOR THAT ORDER CHANGED, and saying so is the point. It used to read that memory -// fit was DECIDABLE from weights, KV cost and capacity alone, so ordering it first turned no -// decidable answer into an unanswerable one. That is no longer true and was never true: fit is now -// established from an observation, so the memory arm can itself return Unanswerable, and the order -// buys refusal economy rather than decidability. A candidate with no qualifying footprint reading is -// unresolved on the FIRST axis, not admitted past it -- which is why the roster answer for a -// partially measured candidate is a refusal to select and not a default win for whoever was -// measured. Only once a candidate is evidenced to be resident do we ask whether it is evidenced to -// RETRIEVE and to SERVE at this floor. +// THE JUSTIFICATION FOR THAT ORDER CHANGED TWICE, and both changes are recorded because each one +// removed a claim that had been doing work. It first read that memory fit was DECIDABLE from +// weights, KV cost and capacity alone. Then it read that fit was established by comparing an +// observed footprint against allocatable memory. NEITHER SURVIVED: the first died when the KV +// constant was falsified, the second when the observed quantity turned out to be a runner BUFFER +// figure that cannot be subtracted from an OS memory total. +// +// SO FIT IS NOT COMPUTED AND IT IS NOT COMPARED -- IT IS EXECUTED. The question the instrument can +// actually answer is whether this realization, at this context, at this concurrency, ON THIS NODE, +// loaded and produced a token. A successful generation is a positive fit under exactly the +// conditions that held; a failed one is an observed non-fit at that configuration; no observation +// at all is Unanswerable. Nothing in this arm subtracts a reported buffer from a declared capacity, +// because those are different accounting domains and the difference between them is not a fact. +// +// The node capacity carrier and its allocatable derivation LEFT THIS MODULE with that comparison. +// They were not kept "for reference": a value the decision cannot consume is decoration, and the +// three-grain distinction they encoded belongs to a capacity authority rather than to a selector +// that can no longer compare against it. +// +// A candidate with no qualifying observation is unresolved on the FIRST axis, not admitted past it +// -- which is why the roster answer for a partially measured candidate is a refusal to select and +// not a default win for whoever was measured. Only once a candidate is evidenced to LOAD do we ask +// whether it is evidenced to RETRIEVE and to SERVE at this floor. // // The context test consumes semantic_context_verified_to -- an observed retrieval depth -- and NEVER // declared_context. A publisher declaring 1,048,576 positions has made a claim about what the @@ -344,35 +354,29 @@ type CandidateVerdict // error the publication layer exists to prevent, one module downstream of where it was removed. fn evaluate_candidate( candidate: QuantizedCandidate, - capacity: NodeCapacity, + node: NonEmptyStr, constraints: ServingConstraints, ) -> CandidateVerdict { - match allocatable(capacity: capacity) { - Absent => ServingCandidateUnanswerable { - identity: candidate.identity, - quant_label: candidate.quant_label, - missing: CapacityMalformed, - } - Present { value: room } => match resident_footprint_at_floor( - observations: candidate.resident_footprint_observations, + match runtime_memory_observation_at_floor( + observations: candidate.runtime_memory_observations, floor: constraints.context_floor, hot_sessions: constraints.hot_sessions, - ) { - Absent => ServingCandidateUnanswerable { + node: node, + ) { + Absent => ServingCandidateUnanswerable { identity: candidate.identity, quant_label: candidate.quant_label, - missing: ResidentFootprintUnmeasuredAtFloor { + missing: MemoryFitUnobservedAtConfiguration { floor: constraints.context_floor, sessions: constraints.hot_sessions, }, } Present { value: observed } => { - let required = observed.resident - match byte_size_count(b: required) > byte_size_count(b: room) { + match !observed.generation_succeeded { true => ServingCandidateRejected { identity: candidate.identity, quant_label: candidate.quant_label, - axis: DoesNotFitMemory { required: required, allocatable: room }, + axis: DoesNotFitMemory { observed: observed }, } false => match candidate.semantic_context_verified_to { Absent => ServingCandidateUnanswerable { @@ -412,8 +416,7 @@ fn evaluate_candidate( } false => ServingCandidateAdmissible { candidate: candidate, - resident: required, - headroom: byte_size(count: byte_size_count(b: room) - byte_size_count(b: required)), + fit: observed, } } } @@ -421,13 +424,12 @@ fn evaluate_candidate( } } } - } } } fn verdict_is_admissible(verdict: CandidateVerdict) -> Bool { match verdict { - ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => true + ServingCandidateAdmissible { candidate: _, fit: _ } => true ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => false } @@ -458,7 +460,7 @@ fn compare_within_release(a: WithinReleaseQuality, b: WithinReleaseQuality) -> I // it. A fact that was never established is carried as absent, not as a sentinel. fn verdict_quality(verdict: CandidateVerdict) -> WithinReleaseQuality? { match verdict { - ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => + ServingCandidateAdmissible { candidate: c, fit: _ } => Present { value: WithinReleaseQuality { identity: c.identity, rank: c.quality_rank } } ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => none ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => none @@ -503,11 +505,11 @@ fn unanswerable_cause_wire(cause: UnanswerableCause) -> String { fn choose_serving_candidate( candidates: List, - capacity: NodeCapacity, + node: NonEmptyStr, constraints: ServingConstraints, ) -> ServingChoice { let verdicts = map(candidates, c => - evaluate_candidate(candidate: c, capacity: capacity, constraints: constraints)) + evaluate_candidate(candidate: c, node: node, constraints: constraints)) let unresolved = filter(verdicts, v => verdict_is_unanswerable(verdict: v)) let admissible = filter(verdicts, v => verdict_is_admissible(verdict: v)) match length(unresolved) > 0 { @@ -543,7 +545,7 @@ fn choose_serving_candidate( fn verdict_is_unanswerable(verdict: CandidateVerdict) -> Bool { match verdict { ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => true - ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false + ServingCandidateAdmissible { candidate: _, fit: _ } => false ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false } } @@ -571,7 +573,7 @@ fn distinct_release_count(verdicts: List) -> Nat { fn verdict_release(verdict: CandidateVerdict) -> ReleaseIdentity { match verdict { - ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => c.identity + ServingCandidateAdmissible { candidate: c, fit: _ } => c.identity ServingCandidateRejected { identity: r, quant_label: _, axis: _ } => r ServingCandidateUnanswerable { identity: r, quant_label: _, missing: _ } => r } diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index 66c28b2f907..a16db8e519c 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -11,16 +11,15 @@ import gunbc.model.publication { ReleaseIdentity } import gunbc.model.choice { PrefillRateUnmeasuredAtFloor, SemanticContextUnverifiedAtFloor, FreshPrefillObservation, prefill_rate_at_floor, - ResidentFootprintObservation, resident_footprint_at_floor, + OllamaRunnerMemoryObservation, runtime_memory_observation_at_floor, FreshSessionPrefill, WarmContinuation, serving_regime_wire, - NodeCapacity, QuantizedCandidate, ServingConstraints, + QuantizedCandidate, ServingConstraints, CandidateVerdict, ServingCandidateAdmissible, ServingCandidateRejected, RejectionAxis, DoesNotFitMemory, ContextBelowFloor, PrefillBelowFloor, ServingChoice, ChoseCandidate, NoCandidateAdmissible, SelectionUnanswerable, UnanswerableCause, UnresolvedCandidateCouldWin, CrossReleaseQualityOrderAbsent, QualityRankTie, - ServingCandidateUnanswerable, MissingFact, ResidentFootprintUnmeasuredAtFloor, CapacityMalformed, + ServingCandidateUnanswerable, MissingFact, MemoryFitUnobservedAtConfiguration, evaluate_candidate, choose_serving_candidate, - allocatable, firmware_carveout, } // THE FIXTURE IS MEASURED, not invented. Every number here was read off the running nodes or @@ -36,29 +35,37 @@ import gunbc.model.choice { // lengths of 512, so a token costs 43 * 1 * (512+512) * 2 = 88,064 bytes." That number is DERIVED // from architecture and was labelled measured, in the fixture for a module whose subject is refusing // exactly that substitution. Loading the build and reading it back falsified both the constant and -// the linear form it was multiplied through (see gunbc.model.choice ResidentFootprintObservation). +// the linear form it was multiplied through (see gunbc.model.choice OllamaRunnerMemoryObservation). // // Every footprint below is now a reading. Re-derive any of them by loading the build at an explicit // num_ctx and reading `size_vram` back from the Ollama /api/ps surface on the named node; that // instrument, not any number transcribed here, is the authority. -data node_capacity: NodeCapacity = NodeCapacity { - node_name: "spark-a3ee" as NonEmptyStr, - nominal: byte_size(count: 137438953472), - kernel_visible: byte_size(count: 130660151296), - runtime_overhead: byte_size(count: 3000000000), -} +// THE NODE THE OBSERVATIONS WERE TAKEN ON. A NodeCapacity carrier stood here, with the vendor +// nominal, the kernel-visible total and a declared runtime overhead. It is gone with the comparison +// that consumed it -- see gunbc.model.choice. Fit is now established by whether a configuration +// LOADED on a named node, so what the selector needs from the host is its identity and not its +// arithmetic. +data serving_node: NonEmptyStr = "spark-a3ee" as NonEmptyStr + +// The declared ordering fixtures live on their own node, so a constraint about the real hosts can +// never be answered by one of them. Node equality is a qualification in the selector, which makes +// that separation structural rather than a naming convention. +data fixture_node: NonEmptyStr = "fixture-node" as NonEmptyStr // THE READINGS, spark-a3ee, Ollama 0.32.9, one resident session, /api/ps size_vram. `size` and // `size_vram` agreed exactly on this unified-memory node, so nothing turns on which field is read. -fn iq2_xxs_footprint(depth: Nat, resident: Nat) -> ResidentFootprintObservation { - ResidentFootprintObservation { +fn iq2_xxs_footprint(depth: Nat, resident: Nat) -> OllamaRunnerMemoryObservation { + OllamaRunnerMemoryObservation { realization: "hf.co/antirez/deepseek-v4-gguf:latest" as NonEmptyStr, runtime: "ollama 0.32.9" as NonEmptyStr, + runtime_config: "OLLAMA_NUM_PARALLEL unset; explicit num_ctx per request" as NonEmptyStr, node: "spark-a3ee" as NonEmptyStr, - instrument: "/api/ps size_vram after load at explicit num_ctx" as NonEmptyStr, + instrument: "/api/ps size and size_vram after load at explicit num_ctx" as NonEmptyStr, context_depth: token_count(count: depth), concurrent_sessions: 1, - resident: byte_size(count: resident), + reported_total_buffer: byte_size(count: resident), + reported_gpu_buffer: byte_size(count: resident), + generation_succeeded: true, } } @@ -98,7 +105,7 @@ data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { FreshPrefillObservation { depth: token_count(count: 255061), rate: tokens_per_second(count: 197) }, FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, ], - resident_footprint_observations: [ + runtime_memory_observations: [ iq2_xxs_footprint(depth: 131072, resident: 86532465622), iq2_xxs_footprint(depth: 262144, resident: 87064355798), iq2_xxs_footprint(depth: 400000, resident: 87692389907), @@ -113,15 +120,18 @@ data build_iq3_s: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context_verified_to: none, fresh_prefill_observations: [], - resident_footprint_observations: [ - ResidentFootprintObservation { + runtime_memory_observations: [ + OllamaRunnerMemoryObservation { realization: "deepseek-v4-flash:iq3s" as NonEmptyStr, runtime: "ollama 0.32.9" as NonEmptyStr, + runtime_config: "OLLAMA_NUM_PARALLEL unset; explicit num_ctx per request" as NonEmptyStr, node: "spark-a3ee" as NonEmptyStr, - instrument: "/api/ps size_vram after load at explicit num_ctx" as NonEmptyStr, + instrument: "/api/ps size and size_vram after load at explicit num_ctx" as NonEmptyStr, context_depth: token_count(count: 400000), concurrent_sessions: 1, - resident: byte_size(count: 116970000000), + reported_total_buffer: byte_size(count: 116970000000), + reported_gpu_buffer: byte_size(count: 116970000000), + generation_succeeded: true, }, ], } @@ -152,7 +162,7 @@ fn chosen_label(choice: ServingChoice) -> String { NoCandidateAdmissible { rejections: _ } => "none" ChoseCandidate { verdict: v } => match v { - ServingCandidateAdmissible { candidate: c, resident: _, headroom: _ } => c.quant_label as String + ServingCandidateAdmissible { candidate: c, fit: _ } => c.quant_label as String ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => "none" ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => "none" } @@ -161,21 +171,14 @@ fn chosen_label(choice: ServingChoice) -> String { // ============================ THE CLAIMS ============================ -// The three capacity grains are DIFFERENT, and only one of them is a capacity. Quoting the vendor -// figure is what made a 128.1 GB build look feasible. -test fn w_capacity_grains_differ_and_only_kernel_visible_is_spendable() -> Bool { - match firmware_carveout(capacity: node_capacity) { - Absent => false - Present { value: c } => - match allocatable(capacity: node_capacity) { - Absent => false - Present { value: a } => - byte_size_count(b: c) == 6778802176 - && byte_size_count(b: a) == 127660151296 - && byte_size_count(b: a) < byte_size_count(b: node_capacity.nominal) - } - } -} +// THE CAPACITY-GRAINS WITNESS STOOD HERE, asserting that nominal, kernel-visible and allocatable +// are three different quantities and that only kernel-visible is spendable. It is deleted with its +// subject: firmware_carveout and allocatable left gunbc.model.choice when the memory arm stopped +// subtracting a runner buffer figure from an OS memory total. The distinction it defended is real +// and already cost one wrong decision, so it belongs to whatever authority owns node capacity -- +// but a witness whose functions no longer exist in the module under test is not evidence, and +// keeping it alive by importing them from somewhere else would be inventing a consumer. + // THE CLAIM THE MODULE EXISTS FOR, AND THE ONE IT GOT WRONG. // @@ -191,17 +194,16 @@ test fn w_capacity_grains_differ_and_only_kernel_visible_is_spendable() -> Bool // and no prefill observation, so it is UNANSWERABLE on semantic context. Unknown is not qualified, // exactly as unknown is not rejected. test fn w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolved() -> Bool { - match evaluate_candidate(candidate: build_iq3_s, capacity: node_capacity, constraints: floor_400k) { + match evaluate_candidate(candidate: build_iq3_s, node: serving_node, constraints: floor_400k) { ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => match m { SemanticContextUnverifiedAtFloor { floor: f, declared: d } => token_count_value(t: f) == 400000 && token_count_value(t: d) == 1048576 - ResidentFootprintUnmeasuredAtFloor { floor: _, sessions: _ } => false + MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false - CapacityMalformed => false } ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false - ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false + ServingCandidateAdmissible { candidate: _, fit: _ } => false } } @@ -210,7 +212,7 @@ test fn w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolve // selector refuses, and the refusal names the unresolved candidate as the reason. test fn w_the_real_roster_refuses_while_the_higher_rank_is_unresolved() -> Bool { match choose_serving_candidate( - candidates: installed(), capacity: node_capacity, constraints: floor_400k) { + candidates: installed(), node: serving_node, constraints: floor_400k) { SelectionUnanswerable { cause: _, evaluations: e } => length(e) == 2 ChoseCandidate { verdict: _ } => false NoCandidateAdmissible { rejections: _ } => false @@ -221,14 +223,17 @@ test fn w_the_real_roster_refuses_while_the_higher_rank_is_unresolved() -> Bool // so the refusal above is not a footprint carrier that answers nothing. Its qualifying reading is // the 400,000 one rather than the 1,048,576 one -- the tightest sound bound at or beyond the floor. test fn w_the_two_bit_build_is_resident_established_at_the_floor() -> Bool { - match resident_footprint_at_floor( - observations: build_iq2_xxs.resident_footprint_observations, + match runtime_memory_observation_at_floor( + observations: build_iq2_xxs.runtime_memory_observations, floor: token_count(count: 400000), hot_sessions: 1, + node: serving_node, ) { Present { value: o } => - byte_size_count(b: o.resident) == 87692389907 + byte_size_count(b: o.reported_total_buffer) == 87692389907 + && byte_size_count(b: o.reported_gpu_buffer) == 87692389907 && token_count_value(t: o.context_depth) == 400000 + && o.generation_succeeded Absent => false } } @@ -237,13 +242,13 @@ test fn w_the_two_bit_build_is_resident_established_at_the_floor() -> Bool { // is unanswerable rather than multiplied. Both arms discriminate against the same observation list // that positively answers one hot session at 400k above. test fn w_a_shallow_or_more_concurrent_demand_has_no_qualifying_footprint() -> Bool { - let obs = build_iq2_xxs.resident_footprint_observations - match resident_footprint_at_floor( - observations: obs, floor: token_count(count: 2000000), hot_sessions: 1) { + let obs = build_iq2_xxs.runtime_memory_observations + match runtime_memory_observation_at_floor( + observations: obs, floor: token_count(count: 2000000), hot_sessions: 1, node: serving_node) { Present { value: _ } => false Absent => - match resident_footprint_at_floor( - observations: obs, floor: token_count(count: 400000), hot_sessions: 2) { + match runtime_memory_observation_at_floor( + observations: obs, floor: token_count(count: 400000), hot_sessions: 2, node: serving_node) { Present { value: _ } => false Absent => true } @@ -261,15 +266,22 @@ test fn w_a_shallow_or_more_concurrent_demand_has_no_qualifying_footprint() -> B // SYNTHETIC footprints for the declared ordering fixtures below. These are not fleet readings and // say so in every field: the node is named "fixture", so a real constraint can never be answered by // one, and the ordering claims they support are about the SELECTOR and not about any hardware. -fn fixture_footprint(resident: Nat) -> ResidentFootprintObservation { - ResidentFootprintObservation { +fn fixture_footprint(resident: Nat) -> OllamaRunnerMemoryObservation { + fixture_load(resident: resident, succeeded: true) +} + +fn fixture_load(resident: Nat, succeeded: Bool) -> OllamaRunnerMemoryObservation { + OllamaRunnerMemoryObservation { realization: "fixture" as NonEmptyStr, runtime: "fixture" as NonEmptyStr, - node: "fixture" as NonEmptyStr, + runtime_config: "fixture" as NonEmptyStr, + node: "fixture-node" as NonEmptyStr, instrument: "declared fixture, not a reading" as NonEmptyStr, context_depth: token_count(count: 1048576), concurrent_sessions: 1, - resident: byte_size(count: resident), + reported_total_buffer: byte_size(count: resident), + reported_gpu_buffer: byte_size(count: resident), + generation_succeeded: succeeded, } } @@ -285,7 +297,7 @@ data fixture_low_rank: QuantizedCandidate = QuantizedCandidate { fresh_prefill_observations: [ FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], - resident_footprint_observations: [fixture_footprint(resident: 28000000000)], + runtime_memory_observations: [fixture_footprint(resident: 28000000000)], } data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { @@ -297,16 +309,16 @@ data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { fresh_prefill_observations: [ FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 400) }, ], - resident_footprint_observations: [fixture_footprint(resident: 38000000000)], + runtime_memory_observations: [fixture_footprint(resident: 38000000000)], } test fn w_the_higher_quality_candidate_wins_when_both_are_admissible() -> Bool { chosen_label(choice: choose_serving_candidate( candidates: [fixture_low_rank, fixture_high_rank], - capacity: node_capacity, constraints: floor_8k)) == "HIGH" + node: fixture_node, constraints: floor_8k)) == "HIGH" && chosen_label(choice: choose_serving_candidate( candidates: [fixture_high_rank, fixture_low_rank], - capacity: node_capacity, constraints: floor_8k)) == "HIGH" + node: fixture_node, constraints: floor_8k)) == "HIGH" } // A DECLARATION IS NOT EVIDENCE. The installed IQ3_S declares 1,048,576 positions and has no @@ -314,16 +326,15 @@ test fn w_the_higher_quality_candidate_wins_when_both_are_admissible() -> Bool { // This is the rung error the publication layer removed, re-tested one module downstream where the // selector could have quietly reintroduced it. test fn w_a_declared_context_never_qualifies_without_a_retrieval_receipt() -> Bool { - match evaluate_candidate(candidate: build_iq3_s, capacity: node_capacity, constraints: floor_8k) { + match evaluate_candidate(candidate: build_iq3_s, node: serving_node, constraints: floor_8k) { ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => match m { SemanticContextUnverifiedAtFloor { floor: _, declared: d } => token_count_value(t: d) == 1048576 - ResidentFootprintUnmeasuredAtFloor { floor: _, sessions: _ } => false + MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false - CapacityMalformed => false } - ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false + ServingCandidateAdmissible { candidate: _, fit: _ } => false ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false } } @@ -348,24 +359,37 @@ test fn w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() -> Bool // axis is read rather than argued. A best-effort pick here would reintroduce the silent degradation // the whole exercise removes. // -// IT IS RUN ON A STARVED NODE, NOT ON AN ABSURD FLOOR, and that change is a consequence of the -// footprint repair rather than a convenience. This claim used to ask for an 8,000,000-token floor, -// which under the old arithmetic multiplied out to a memory REJECTION. There is no footprint reading -// at that depth and never will be one, so the honest answer there is now UNANSWERABLE -- the -// candidates have not failed the test, they have not taken it. To witness rejection the candidates -// must be evidenced and the room must be genuinely too small, so the capacity shrinks instead of the -// floor growing. The fixtures are used because they are resident-established at depth by -// construction; the installed builds are not, at any depth beyond what was actually loaded. +// A MEMORY REJECTION IS NOW AN OBSERVED FAILED LOAD, and this claim moved twice to keep saying that +// honestly. It first asked for an 8,000,000-token floor, which under the old arithmetic multiplied +// out to a rejection; there is no reading at that depth, so the honest answer became Unanswerable. +// It then starved a NodeCapacity so the subtraction would go negative -- and that carrier is gone, +// because the subtraction spanned two accounting domains. +// +// What is left is the only thing the instrument can actually report as a non-fit: a configuration +// that was attempted on the node and DID NOT PRODUCE A TOKEN. That is an observation, not a +// calculation, and it is the arm a real out-of-memory load would populate. test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { - let starved = NodeCapacity { - node_name: "fixture-starved" as NonEmptyStr, - nominal: byte_size(count: 34000000000), - kernel_visible: byte_size(count: 30000000000), - runtime_overhead: byte_size(count: 3000000000), + let did_not_load_low = QuantizedCandidate { + identity: fixture_low_rank.identity, + quant_label: fixture_low_rank.quant_label, + quality_rank: fixture_low_rank.quality_rank, + declared_context: fixture_low_rank.declared_context, + semantic_context_verified_to: fixture_low_rank.semantic_context_verified_to, + fresh_prefill_observations: fixture_low_rank.fresh_prefill_observations, + runtime_memory_observations: [fixture_load(resident: 28000000000, succeeded: false)], + } + let did_not_load_high = QuantizedCandidate { + identity: fixture_high_rank.identity, + quant_label: fixture_high_rank.quant_label, + quality_rank: fixture_high_rank.quality_rank, + declared_context: fixture_high_rank.declared_context, + semantic_context_verified_to: fixture_high_rank.semantic_context_verified_to, + fresh_prefill_observations: fixture_high_rank.fresh_prefill_observations, + runtime_memory_observations: [fixture_load(resident: 38000000000, succeeded: false)], } match choose_serving_candidate( - candidates: [fixture_low_rank, fixture_high_rank], - capacity: starved, + candidates: [did_not_load_low, did_not_load_high], + node: fixture_node, constraints: floor_400k, ) { NoCandidateAdmissible { rejections: r } => length(r) == 2 @@ -416,23 +440,21 @@ test fn w_a_warm_floor_over_fresh_only_receipts_is_unanswerable() -> Bool { prefill_floor: tokens_per_second(count: 1), prefill_regime: WarmContinuation, } - match evaluate_candidate(candidate: build_iq2_xxs, capacity: node_capacity, constraints: warm_400k) { + match evaluate_candidate(candidate: build_iq2_xxs, node: serving_node, constraints: warm_400k) { ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => match m { PrefillRateUnmeasuredAtFloor { floor: _, regime: g } => serving_regime_wire(regime: g) == "warm-continuation" - ResidentFootprintUnmeasuredAtFloor { floor: _, sessions: _ } => false + MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false - CapacityMalformed => false } - ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false + ServingCandidateAdmissible { candidate: _, fit: _ } => false ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false } } test fn w_all_serving_choice_claims_hold() -> Bool { - w_capacity_grains_differ_and_only_kernel_visible_is_spendable() - && w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolved() + w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolved() && w_the_real_roster_refuses_while_the_higher_rank_is_unresolved() && w_the_two_bit_build_is_resident_established_at_the_floor() && w_a_shallow_or_more_concurrent_demand_has_no_qualifying_footprint() @@ -460,21 +482,20 @@ data build_unmeasured: QuantizedCandidate = QuantizedCandidate { fresh_prefill_observations: [ FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, ], - resident_footprint_observations: [], + runtime_memory_observations: [], } test fn w_an_unmeasured_candidate_is_unanswerable_not_rejected() -> Bool { - match evaluate_candidate(candidate: build_unmeasured, capacity: node_capacity, constraints: floor_400k) { + match evaluate_candidate(candidate: build_unmeasured, node: serving_node, constraints: floor_400k) { ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => match m { - ResidentFootprintUnmeasuredAtFloor { floor: f, sessions: n } => + MemoryFitUnobservedAtConfiguration { floor: f, sessions: n } => token_count_value(t: f) == 400000 && n == 1 PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false - CapacityMalformed => false } ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false - ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false + ServingCandidateAdmissible { candidate: _, fit: _ } => false } } @@ -482,7 +503,7 @@ test fn w_an_unmeasured_candidate_is_unanswerable_not_rejected() -> Bool { // outstanding: choosing the best ANSWERABLE candidate is not choosing the best candidate. test fn w_selection_refuses_while_an_unresolved_candidate_could_win() -> Bool { match choose_serving_candidate( - candidates: [build_iq2_xxs, build_unmeasured], capacity: node_capacity, constraints: floor_400k) { + candidates: [build_iq2_xxs, build_unmeasured], node: serving_node, constraints: floor_400k) { SelectionUnanswerable { cause: c, evaluations: _ } => match c { UnresolvedCandidateCouldWin { unresolved_count: n } => n == 1 @@ -505,12 +526,25 @@ data other_release: QuantizedCandidate = QuantizedCandidate { fresh_prefill_observations: [ FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 900) }, ], - resident_footprint_observations: [fixture_footprint(resident: 30600000000)], + runtime_memory_observations: [ + OllamaRunnerMemoryObservation { + realization: "qwen3.6:35b" as NonEmptyStr, + runtime: "ollama 0.32.9" as NonEmptyStr, + runtime_config: "declared fixture, not a reading" as NonEmptyStr, + node: "spark-a3ee" as NonEmptyStr, + instrument: "declared fixture, not a reading" as NonEmptyStr, + context_depth: token_count(count: 1048576), + concurrent_sessions: 1, + reported_total_buffer: byte_size(count: 30600000000), + reported_gpu_buffer: byte_size(count: 30600000000), + generation_succeeded: true, + }, + ], } test fn w_two_admissible_releases_refuse_for_want_of_a_cross_release_order() -> Bool { match choose_serving_candidate( - candidates: [build_iq2_xxs, other_release], capacity: node_capacity, constraints: floor_400k) { + candidates: [build_iq2_xxs, other_release], node: serving_node, constraints: floor_400k) { SelectionUnanswerable { cause: c, evaluations: _ } => match c { CrossReleaseQualityOrderAbsent { distinct_release_count: n } => n == 2 @@ -522,39 +556,23 @@ test fn w_two_admissible_releases_refuse_for_want_of_a_cross_release_order() -> } } -// A malformed capacity is a fact about the INPUTS, not a verdict about a candidate. Clamping the +// THE MALFORMED-CAPACITY WITNESS STOOD HERE. It asserted that an overhead allowance exceeding the +// machine is a fact about the INPUTS rather than a verdict about a candidate, and that clamping the // Nat subtraction to zero would turn an impossible machine into a machine with no room and reject -// every candidate on memory for a reason that was never true. -data malformed_capacity: NodeCapacity = NodeCapacity { - node_name: "impossible" as NonEmptyStr, - nominal: byte_size(count: 137438953472), - kernel_visible: byte_size(count: 130660151296), - runtime_overhead: byte_size(count: 200000000000), -} +// every candidate for a reason that was never true. That reasoning is still right, and it is why the +// subtraction answered with an option rather than trapping. +// +// It is deleted because the subtraction is gone. CapacityMalformed had exactly one producer, and +// once the memory arm stopped comparing a runner buffer figure against an OS memory total there was +// no arithmetic left to malform. Retaining the witness would have meant retaining the carrier to +// give it something to test, which is a check keeping its own subject alive. -test fn w_malformed_capacity_is_unanswerable_not_a_memory_rejection() -> Bool { - match allocatable(capacity: malformed_capacity) { - Present { value: _ } => false - Absent => - match evaluate_candidate(candidate: build_iq2_xxs, capacity: malformed_capacity, constraints: floor_400k) { - ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => - match m { - CapacityMalformed => true - ResidentFootprintUnmeasuredAtFloor { floor: _, sessions: _ } => false - PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false - SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false - } - ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false - ServingCandidateAdmissible { candidate: _, resident: _, headroom: _ } => false - } - } -} test fn w_all_answerability_claims_hold() -> Bool { w_an_unmeasured_candidate_is_unanswerable_not_rejected() && w_selection_refuses_while_an_unresolved_candidate_could_win() && w_two_admissible_releases_refuse_for_want_of_a_cross_release_order() - && w_malformed_capacity_is_unanswerable_not_a_memory_rejection() + } // A TIE HAS NO MAXIMUM. Two admissible candidates of the SAME release at the SAME rank must refuse @@ -570,7 +588,7 @@ data fixture_tied_a: QuantizedCandidate = QuantizedCandidate { fresh_prefill_observations: [ FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], - resident_footprint_observations: [fixture_footprint(resident: 28000000000)], + runtime_memory_observations: [fixture_footprint(resident: 28000000000)], } data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { @@ -582,7 +600,7 @@ data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { fresh_prefill_observations: [ FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], - resident_footprint_observations: [fixture_footprint(resident: 29000000000)], + runtime_memory_observations: [fixture_footprint(resident: 29000000000)], } fn is_tie_refusal(choice: ServingChoice) -> Bool { @@ -600,7 +618,7 @@ fn is_tie_refusal(choice: ServingChoice) -> Bool { test fn w_tied_quality_ranks_refuse_in_both_roster_orders() -> Bool { is_tie_refusal(choice: choose_serving_candidate( - candidates: [fixture_tied_a, fixture_tied_b], capacity: node_capacity, constraints: floor_8k)) + candidates: [fixture_tied_a, fixture_tied_b], node: fixture_node, constraints: floor_8k)) && is_tie_refusal(choice: choose_serving_candidate( - candidates: [fixture_tied_b, fixture_tied_a], capacity: node_capacity, constraints: floor_8k)) + candidates: [fixture_tied_b, fixture_tied_a], node: fixture_node, constraints: floor_8k)) } From eacbb2f7523c8fd0df7f9dab8a3a6ae7abe691ea Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 21:29:37 +0000 Subject: [PATCH 14/22] Join every evidence axis to one realization identity, and type what an attempt produced Findings 1-7 of the side chat's REQUEST_CHANGES. (1) Memory, retrieval and prefill evidence now join a ServingRealizationIdentity -- release x quantization x artifact x runtime x runtime-config. The fields were carried and never matched, so evidence from another artifact, runtime or configuration qualified unchanged, and memory fit from A plus retrieval from B plus prefill from C could be admitted as one candidate that never existed. Negative controls flip each component of the key in turn, plus the node, and require the evidence to stop answering; the positive control keeps them from being satisfied by a filter that rejects everything. (2) An unsuccessful attempt is no longer a memory verdict. RunnerAttemptOutcome splits served / refused-for-memory / failed-for-other-cause, and the reported buffers live only on the arm where a runner existed to report them -- so a pre-runner failure carrying byte figures is unwritable rather than discouraged. Only the typed insufficient-memory refusal rejects; every other failure is Unanswerable and names its cause. (3) Magnitude no longer decides between contradictory readings. The fold picked the smallest reported buffer BEFORE reading the success flag, so a small failed receipt beat a large successful one. Qualification is now an identity join first, and two qualifying attempts that disagree refuse -- exercised in both roster orders, with the large-buffer success alone still admitting. (4) The real-roster claim asserts the exact cause (UnresolvedCandidateCouldWin, unresolved_count == 1) and the exact verdict arms, not just the evaluation count. (5) The IQ3_S non-fit rationale is withdrawn from population.dag -- loading the build falsified it. The finer split the measurements DO establish replaces it, cited by naming the witness rather than transcribing verdicts. (6) Completeness is publisher-scoped. Answering Answerable for any publisher catalogue said one publisher's index enumerates the open-weight universe: a real count over the wrong population, the original error re-committed by the function written to prevent it. The universe-wide question has no answerable arm. (7) The population subset claim is rung-honest. narrow_population constructs a subset and now retains the parent roster and refuses backward or same-stage transitions, but ModelPopulation is an ordinary record -- the class is mechanically preventable, not structurally impossible, and its trigger is constructor privacy in the substrate. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01U397y4s3dSBof7vGPAX87G --- dag/gunbc/model/choice.dag | 409 +++++++++---- dag/gunbc/model/population.dag | 197 +++++-- ...odel_population_narrowing_witness_test.dag | 110 +++- .../model/serving_choice_witness_test.dag | 549 ++++++++++++++---- 4 files changed, 967 insertions(+), 298 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index 682bc1ede2a..9cd9362d7f4 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -59,16 +59,54 @@ import std.measure { // numbers carry no unit and must never be compared across releases, because a Q3 of a 284B model and // a Q8 of a 30B model are not on one scale. Cross-release quality is the operator's subjective call, // which is why it enters as a floor rather than as an objective. -type QuantizedCandidate { - identity: ReleaseIdentity +// THE EXACT THING EVIDENCE IS ABOUT. A release is not what serves; a release quantized into an +// artifact, loaded by a particular runtime under a particular configuration, is. +// +// WHY THIS EXISTS AND WHY IT IS A JOIN KEY RATHER THAN A LABEL. The observations below used to +// CARRY realization, runtime and configuration as fields and never MATCH on them: qualification +// filtered on depth, concurrency and node alone. An observation for another artifact, another +// runtime, or another configuration could be placed in a candidate's list and qualify unchanged, and +// every provenance string could be rewritten without a single claim going red. Fields that describe +// a qualification without performing it are the same failure as a constant labelled "exact" -- they +// read as evidence and do no work. +// +// Worse than any single mismatch is the COMBINATION it permitted: memory fit from configuration A, +// retrieval evidence from B, prefill evidence from C, admitted as one candidate that never existed. +// So every evidence axis joins THIS identity, and an observation that does not carry it cannot +// answer for this candidate at all. +type ServingRealizationIdentity { + release_id: ReleaseIdentity quant_label: NonEmptyStr + artifact_ref: NonEmptyStr + runtime: NonEmptyStr + runtime_config: NonEmptyStr +} + +fn serving_realization_equal(a: ServingRealizationIdentity, b: ServingRealizationIdentity) -> Bool { + release_identity_equal(a: a.release_id, b: b.release_id) + && (a.quant_label as String) == (b.quant_label as String) + && (a.artifact_ref as String) == (b.artifact_ref as String) + && (a.runtime as String) == (b.runtime as String) + && (a.runtime_config as String) == (b.runtime_config as String) +} + +type QuantizedCandidate { + realization: ServingRealizationIdentity quality_rank: Int declared_context: TokenCount - semantic_context_verified_to: TokenCount? + semantic_context: SemanticContextEvidence? fresh_prefill_observations: List runtime_memory_observations: List } +// Retrieval evidence is an OBSERVATION about a realization, not a bare depth hanging off the +// candidate. It was a `TokenCount?` -- a number with no subject, which could not be checked against +// the thing it was being used to qualify. +type SemanticContextEvidence { + realization: ServingRealizationIdentity + verified_to: TokenCount +} + // PREFILL RATE IS NOT A SCALAR, and this repository's own measurements falsified the scalar that // used to sit here. The same realization measured 253 tok/s at 160,060 tokens, 197 at 255,061 and // 135 at 400,060: attention cost grows with depth, so a single rate silently describes whichever @@ -112,6 +150,7 @@ fn serving_regime_wire(regime: ServingRegime) -> String { // NEXT-RUNG TRIGGER: a PrefixReuseReceipt carrier produced by the observation layer, at which point // a WarmPrefillObservation may be constructed FROM one and not otherwise. type FreshPrefillObservation { + realization: ServingRealizationIdentity depth: TokenCount rate: TokensPerSecond } @@ -128,21 +167,26 @@ type FreshPrefillObservation { // read at all, so the widening arm does not exist to be taken. fn prefill_rate_at_floor( observations: List, + realization: ServingRealizationIdentity, floor: TokenCount, regime: ServingRegime, ) -> TokensPerSecond? { match regime { WarmContinuation => none RestoredSession => none - FreshSessionPrefill => slowest_fresh_rate_at_floor(observations: observations, floor: floor) + FreshSessionPrefill => + slowest_fresh_rate_at_floor(observations: observations, realization: realization, floor: floor) } } fn slowest_fresh_rate_at_floor( observations: List, + realization: ServingRealizationIdentity, floor: TokenCount, ) -> TokensPerSecond? { - let deep = filter(observations, o => token_count_value(t: o.depth) >= token_count_value(t: floor)) + let deep = filter(observations, o => + serving_realization_equal(a: o.realization, b: realization) + && token_count_value(t: o.depth) >= token_count_value(t: floor)) fold(deep, none, (worst, o) => match worst { Absent => Present { value: o.rate } @@ -188,79 +232,146 @@ fn slowest_fresh_rate_at_floor( // produced a token on that node. On the measured unified-memory nodes the two figures were // byte-identical in every reading, but that equality is carried here as data to be observed rather // than assumed, because a full-offload receipt is what would license treating them as one. +// WHAT AN ATTEMPT PRODUCED, as a typed outcome rather than a Bool beside two byte fields. +// +// A Bool `generation_succeeded` said too much and too little at once. Too much, because it mapped +// EVERY unsuccessful attempt to a memory verdict, when a token can fail to appear from request +// validation, tokenization, a template error, an unsupported capability, cancellation, a timeout, a +// transport fault, or a crash that has nothing to do with memory. Too little, because the record +// still carried two buffer figures -- and an attempt that fails BEFORE a runner exists has no +// /api/ps row to read, so those numbers could only be invented. The fixture proved the point by +// inventing them. +// +// Buffer quantities now live only on the arm where a runner existed to report them, so a failed +// pre-runner attempt carrying byte figures is unwritable rather than merely discouraged. Only a +// typed insufficient-memory refusal may become a memory rejection; every other failure is a +// different subject and is reported as one. +type RunnerAttemptOutcome + = RunnerServed { reported_total_buffer: ByteSize, reported_gpu_buffer: ByteSize } + | RunnerRefusedForMemory { detail: NonEmptyStr } + | RunnerFailedForOtherCause { cause: NonEmptyStr } + type OllamaRunnerMemoryObservation { - realization: NonEmptyStr - runtime: NonEmptyStr - runtime_config: NonEmptyStr + realization: ServingRealizationIdentity node: NonEmptyStr instrument: NonEmptyStr context_depth: TokenCount concurrent_sessions: Nat - reported_total_buffer: ByteSize - reported_gpu_buffer: ByteSize - generation_succeeded: Bool -} - -// The observation that answers a demand for `floor` context at `hot_sessions` resident sessions ON -// THIS NODE, or Absent when none does. Node equality is a qualification and not a label: a fit -// established on one machine says nothing about another, and importing it would be the same -// substitution as importing a shallower depth. -// -// DEEPER QUALIFIES SHALLOWER; SHALLOWER NEVER QUALIFIES DEEPER. That direction is the single -// assumption this function makes, and it is stated rather than buried: footprint is monotone in -// depth, so a reading at or beyond the floor upper-bounds the floor cost, while a shallower reading -// bounds nothing and flatters the candidate. Among qualifying readings the SMALLEST is the tightest -// sound bound, so that is the one returned. -// -// CONCURRENCY IS NOT RECOVERED BY MULTIPLICATION. A one-session reading says nothing executable -// about two simultaneously resident sessions -- shared weights, allocator behaviour and per-session -// caches do not decompose into a per-session term this instrument can see. A demand above the -// observed concurrency is UNANSWERABLE, exactly as a deeper floor is, and never a one-session number -// scaled up. -// -// THE WHOLE OBSERVATION IS RETURNED, not a byte count. Detaching a number would strip the depth, -// node, runtime and instrument that make the answer checkable, and a fit established from a reading -// nobody can locate again is not established. -// -// SELECTION AMONG QUALIFYING OBSERVATIONS IS BY REPORTED BUFFER AND NOT BY ROSTER ORDER. The -// comparison is between two readings from ONE instrument, so it stays inside a single accounting -// domain -- which is exactly the property the discarded capacity comparison did not have. -fn runtime_memory_observation_at_floor( + outcome: RunnerAttemptOutcome +} + +// WHAT THE QUALIFYING ATTEMPTS SAY, reconciled by TYPE rather than by byte magnitude. +// +// The previous version folded every qualifying observation to the smallest reported buffer and only +// then looked at whether the attempt succeeded. That let magnitude decide truth: a small failed +// receipt beat a large successful one and flipped admission into rejection, and the reverse pairing +// admitted a candidate whose deeper attempt had failed. Once bytes are no longer compared against a +// capacity, "smallest buffer" is not evidence of fit and was never this selector's objective. +// +// Qualification is an identity join FIRST -- realization, node, depth, concurrency -- and only +// attempts that pass it are reconciled. Contradiction between them REFUSES rather than being settled +// by a number: two attempts at the same qualifying configuration that disagree about whether a +// runner could be built are not a measurement, they are a question. +type MemoryFitEvidence + = MemoryFitEstablished { by: OllamaRunnerMemoryObservation } + | MemoryFitRefusedForMemory { by: OllamaRunnerMemoryObservation } + | MemoryFitContradicted { served: OllamaRunnerMemoryObservation, refused: OllamaRunnerMemoryObservation } + | MemoryFitAttemptFailedOtherwise { by: OllamaRunnerMemoryObservation } + | MemoryFitUnobserved + +fn qualifying_memory_observations( observations: List, + realization: ServingRealizationIdentity, floor: TokenCount, hot_sessions: Nat, node: NonEmptyStr, -) -> OllamaRunnerMemoryObservation? { - let qualifying = filter(observations, o => - token_count_value(t: o.context_depth) >= token_count_value(t: floor) +) -> List { + filter(observations, o => + serving_realization_equal(a: o.realization, b: realization) + && token_count_value(t: o.context_depth) >= token_count_value(t: floor) && o.concurrent_sessions >= hot_sessions && (o.node as String) == (node as String)) - let tightest: OllamaRunnerMemoryObservation? = none - fold(qualifying, tightest, (best, o) => - match best { - Absent => Present { value: o } - Present { value: incumbent } => Present { value: tighter_bound(a: o, b: incumbent) } - }) } -// The comparison is a NAMED BINARY OPERATION rather than an inline match inside the fold, and the -// reason is a compiler limitation worth recording rather than routing silently around: a value bound -// by a `Present { value: x }` pattern over a fold accumulator does not carry its type to a FIELD -// ACCESS -- `x.resident` resolves as `no field 'resident' on type 'Unit'` -- while the same binding -// passed as a typed function ARGUMENT unifies correctly, which is why the prefill fold above never -// hit it. Seeding the fold with an annotated `none` does not repair the inference. So the accumulator -// type is established here by the parameter list. This is a real gap in accumulator type propagation, -// not a stylistic preference, and it will keep shaping folds until it is fixed at the language layer. -fn tighter_bound( +fn outcome_is_served(o: OllamaRunnerMemoryObservation) -> Bool { + match o.outcome { + RunnerServed { reported_total_buffer: _, reported_gpu_buffer: _ } => true + RunnerRefusedForMemory { detail: _ } => false + RunnerFailedForOtherCause { cause: _ } => false + } +} + +fn outcome_is_memory_refusal(o: OllamaRunnerMemoryObservation) -> Bool { + match o.outcome { + RunnerServed { reported_total_buffer: _, reported_gpu_buffer: _ } => false + RunnerRefusedForMemory { detail: _ } => true + RunnerFailedForOtherCause { cause: _ } => false + } +} + +fn outcome_is_other_failure(o: OllamaRunnerMemoryObservation) -> Bool { + match o.outcome { + RunnerServed { reported_total_buffer: _, reported_gpu_buffer: _ } => false + RunnerRefusedForMemory { detail: _ } => false + RunnerFailedForOtherCause { cause: _ } => true + } +} + +// The SHALLOWEST qualifying attempt, because it is the one closest to the configuration actually +// requested. Depth is the axis the demand is expressed in, so ordering on it is ordering on the +// question rather than on an incidental byte count. +fn shallowest( a: OllamaRunnerMemoryObservation, b: OllamaRunnerMemoryObservation, ) -> OllamaRunnerMemoryObservation { - match byte_size_count(b: a.reported_total_buffer) < byte_size_count(b: b.reported_total_buffer) { + match token_count_value(t: a.context_depth) < token_count_value(t: b.context_depth) { true => a false => b } } +fn shallowest_of( + observations: List, +) -> OllamaRunnerMemoryObservation? { + let seed: OllamaRunnerMemoryObservation? = none + fold(observations, seed, (best, o) => + match best { + Absent => Present { value: o } + Present { value: incumbent } => Present { value: shallowest(a: o, b: incumbent) } + }) +} + +fn memory_fit_evidence( + observations: List, + realization: ServingRealizationIdentity, + floor: TokenCount, + hot_sessions: Nat, + node: NonEmptyStr, +) -> MemoryFitEvidence { + let qualifying = qualifying_memory_observations( + observations: observations, realization: realization, + floor: floor, hot_sessions: hot_sessions, node: node) + let served = filter(qualifying, o => outcome_is_served(o: o)) + let refused = filter(qualifying, o => outcome_is_memory_refusal(o: o)) + let broken = filter(qualifying, o => outcome_is_other_failure(o: o)) + match shallowest_of(observations: served) { + Present { value: s } => + match shallowest_of(observations: refused) { + Present { value: r } => MemoryFitContradicted { served: s, refused: r } + Absent => MemoryFitEstablished { by: s } + } + Absent => + match shallowest_of(observations: refused) { + Present { value: r } => MemoryFitRefusedForMemory { by: r } + Absent => + match shallowest_of(observations: broken) { + Present { value: b } => MemoryFitAttemptFailedOtherwise { by: b } + Absent => MemoryFitUnobserved + } + } + } +} + // ============================ CONSTRAINTS ============================ // The operator's floors. Each is a REFUSAL THRESHOLD, never a preference to be traded away silently: @@ -280,6 +391,8 @@ type ServingConstraints { // and the answer would look identical to one where the better candidate genuinely lost. type MissingFact = MemoryFitUnobservedAtConfiguration { floor: TokenCount, sessions: Nat } + | MemoryFitContradictedAtConfiguration { floor: TokenCount, sessions: Nat } + | MemoryAttemptFailedForNonMemoryCause { cause: NonEmptyStr } | PrefillRateUnmeasuredAtFloor { floor: TokenCount, regime: ServingRegime } | SemanticContextUnverifiedAtFloor { floor: TokenCount, declared: TokenCount } @@ -290,6 +403,18 @@ fn missing_fact_wire(missing: MissingFact) -> String { "no runner-memory observation on this node at or beyond ", to_string(token_count_value(t: f)), " tokens with at least ", to_string(n), " concurrent session(s)", ], "") + MemoryFitContradictedAtConfiguration { floor: f, sessions: n } => + join([ + "qualifying runner attempts on this node at or beyond ", to_string(token_count_value(t: f)), + " tokens with at least ", to_string(n), + " concurrent session(s) DISAGREE: one served and one refused for memory at the same joined ", + "configuration, so no memory verdict is available without reconciling them", + ], "") + MemoryAttemptFailedForNonMemoryCause { cause: c } => + join([ + "the only qualifying runner attempt failed for a cause that is not memory (", + (c as String), "), which decides nothing about whether this realization fits", + ], "") PrefillRateUnmeasuredAtFloor { floor: f, regime: g } => join([ "no ", serving_regime_wire(regime: g), " observation at or beyond ", @@ -332,10 +457,16 @@ type CandidateVerdict // // SO FIT IS NOT COMPUTED AND IT IS NOT COMPARED -- IT IS EXECUTED. The question the instrument can // actually answer is whether this realization, at this context, at this concurrency, ON THIS NODE, -// loaded and produced a token. A successful generation is a positive fit under exactly the -// conditions that held; a failed one is an observed non-fit at that configuration; no observation -// at all is Unanswerable. Nothing in this arm subtracts a reported buffer from a declared capacity, -// because those are different accounting domains and the difference between them is not a fact. +// loaded and produced a token. Nothing in this arm subtracts a reported buffer from a declared +// capacity, because those are different accounting domains and the difference between them is not +// a fact. +// +// AN UNSUCCESSFUL ATTEMPT IS NOT A MEMORY VERDICT, and reading it as one was the third error on this +// axis. Only a typed insufficient-memory refusal rejects; any other failure leaves the memory +// question unanswered and says so, because a template error or a cancelled request is evidence about +// the request, not about whether the weights and cache fit. No observation at all is Unanswerable, +// and two qualifying observations that disagree are Unanswerable too -- a disagreement is a question, +// and settling it by preferring whichever number is smaller would let magnitude decide truth. // // The node capacity carrier and its allocatable derivation LEFT THIS MODULE with that comparison. // They were not kept "for reference": a value the decision cannot consume is decoration, and the @@ -352,77 +483,109 @@ type CandidateVerdict // architecture admits, not a measurement of what the model retrieves, and the gap between them is // exactly where silent degradation lives. Admitting on the declaration would reintroduce the rung // error the publication layer exists to prevent, one module downstream of where it was removed. +fn unanswerable(candidate: QuantizedCandidate, missing: MissingFact) -> CandidateVerdict { + ServingCandidateUnanswerable { + identity: candidate.realization.release_id, + quant_label: candidate.realization.quant_label, + missing: missing, + } +} + +fn rejected(candidate: QuantizedCandidate, axis: RejectionAxis) -> CandidateVerdict { + ServingCandidateRejected { + identity: candidate.realization.release_id, + quant_label: candidate.realization.quant_label, + axis: axis, + } +} + fn evaluate_candidate( candidate: QuantizedCandidate, node: NonEmptyStr, constraints: ServingConstraints, ) -> CandidateVerdict { - match runtime_memory_observation_at_floor( - observations: candidate.runtime_memory_observations, - floor: constraints.context_floor, - hot_sessions: constraints.hot_sessions, + match memory_fit_evidence( + observations: candidate.runtime_memory_observations, + realization: candidate.realization, + floor: constraints.context_floor, + hot_sessions: constraints.hot_sessions, node: node, ) { - Absent => ServingCandidateUnanswerable { - identity: candidate.identity, - quant_label: candidate.quant_label, - missing: MemoryFitUnobservedAtConfiguration { + MemoryFitUnobserved => unanswerable(candidate: candidate, missing: MemoryFitUnobservedAtConfiguration { + floor: constraints.context_floor, + sessions: constraints.hot_sessions, + }) + MemoryFitContradicted { served: _, refused: _ } => + unanswerable(candidate: candidate, missing: MemoryFitContradictedAtConfiguration { + floor: constraints.context_floor, + sessions: constraints.hot_sessions, + }) + MemoryFitAttemptFailedOtherwise { by: b } => + unanswerable(candidate: candidate, missing: MemoryAttemptFailedForNonMemoryCause { + cause: attempt_failure_cause(observation: b), + }) + MemoryFitRefusedForMemory { by: r } => + rejected(candidate: candidate, axis: DoesNotFitMemory { observed: r }) + MemoryFitEstablished { by: observed } => + match semantic_context_verified_to( + evidence: candidate.semantic_context, + realization: candidate.realization, + ) { + Absent => unanswerable(candidate: candidate, missing: SemanticContextUnverifiedAtFloor { floor: constraints.context_floor, - sessions: constraints.hot_sessions, - }, - } - Present { value: observed } => { - match !observed.generation_succeeded { - true => ServingCandidateRejected { - identity: candidate.identity, - quant_label: candidate.quant_label, - axis: DoesNotFitMemory { observed: observed }, - } - false => match candidate.semantic_context_verified_to { - Absent => ServingCandidateUnanswerable { - identity: candidate.identity, - quant_label: candidate.quant_label, - missing: SemanticContextUnverifiedAtFloor { + declared: candidate.declared_context, + }) + Present { value: verified } => + match token_count_value(t: verified) < token_count_value(t: constraints.context_floor) { + true => rejected(candidate: candidate, axis: ContextBelowFloor { + declared: verified, + floor: constraints.context_floor, + }) + false => match prefill_rate_at_floor( + observations: candidate.fresh_prefill_observations, + realization: candidate.realization, + floor: constraints.context_floor, + regime: constraints.prefill_regime, + ) { + Absent => unanswerable(candidate: candidate, missing: PrefillRateUnmeasuredAtFloor { floor: constraints.context_floor, - declared: candidate.declared_context, - }, - } - Present { value: verified } => - match token_count_value(t: verified) < token_count_value(t: constraints.context_floor) { - true => ServingCandidateRejected { - identity: candidate.identity, - quant_label: candidate.quant_label, - axis: ContextBelowFloor { declared: verified, floor: constraints.context_floor }, - } - false => match prefill_rate_at_floor( - observations: candidate.fresh_prefill_observations, - floor: constraints.context_floor, - regime: constraints.prefill_regime, - ) { - Absent => ServingCandidateUnanswerable { - identity: candidate.identity, - quant_label: candidate.quant_label, - missing: PrefillRateUnmeasuredAtFloor { - floor: constraints.context_floor, - regime: constraints.prefill_regime, - }, - } - Present { value: prefill } => - match tokens_per_second_count(r: prefill) < tokens_per_second_count(r: constraints.prefill_floor) { - true => ServingCandidateRejected { - identity: candidate.identity, - quant_label: candidate.quant_label, - axis: PrefillBelowFloor { measured: prefill, floor: constraints.prefill_floor }, - } - false => ServingCandidateAdmissible { - candidate: candidate, - fit: observed, - } - } + regime: constraints.prefill_regime, + }) + Present { value: prefill } => + match tokens_per_second_count(r: prefill) < tokens_per_second_count(r: constraints.prefill_floor) { + true => rejected(candidate: candidate, axis: PrefillBelowFloor { + measured: prefill, + floor: constraints.prefill_floor, + }) + false => ServingCandidateAdmissible { candidate: candidate, fit: observed } } - } + } } - } + } + } +} + +fn attempt_failure_cause(observation: OllamaRunnerMemoryObservation) -> NonEmptyStr { + match observation.outcome { + RunnerServed { reported_total_buffer: _, reported_gpu_buffer: _ } => "served" as NonEmptyStr + RunnerRefusedForMemory { detail: d } => d + RunnerFailedForOtherCause { cause: c } => c + } +} + +// The retrieval receipt counts only when it was taken against THIS realization. A depth verified on +// a different quantization or a different artifact is a fact about that realization, and reading it +// here would let a candidate borrow evidence it never earned. +fn semantic_context_verified_to( + evidence: SemanticContextEvidence?, + realization: ServingRealizationIdentity, +) -> TokenCount? { + match evidence { + Absent => none + Present { value: e } => + match serving_realization_equal(a: e.realization, b: realization) { + true => Present { value: e.verified_to } + false => none } } } @@ -461,7 +624,7 @@ fn compare_within_release(a: WithinReleaseQuality, b: WithinReleaseQuality) -> I fn verdict_quality(verdict: CandidateVerdict) -> WithinReleaseQuality? { match verdict { ServingCandidateAdmissible { candidate: c, fit: _ } => - Present { value: WithinReleaseQuality { identity: c.identity, rank: c.quality_rank } } + Present { value: WithinReleaseQuality { identity: c.realization.release_id, rank: c.quality_rank } } ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => none ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => none } @@ -573,7 +736,7 @@ fn distinct_release_count(verdicts: List) -> Nat { fn verdict_release(verdict: CandidateVerdict) -> ReleaseIdentity { match verdict { - ServingCandidateAdmissible { candidate: c, fit: _ } => c.identity + ServingCandidateAdmissible { candidate: c, fit: _ } => c.realization.release_id ServingCandidateRejected { identity: r, quant_label: _, axis: _ } => r ServingCandidateUnanswerable { identity: r, quant_label: _, missing: _ } => r } diff --git a/dag/gunbc/model/population.dag b/dag/gunbc/model/population.dag index 801bbfb1766..6f5d508594e 100644 --- a/dag/gunbc/model/population.dag +++ b/dag/gunbc/model/population.dag @@ -16,25 +16,37 @@ import gunbc.model.publication { // about the release; whether a particular artifact loads, fits or answers well is a verdict about a // REALIZATION, and no such verdict is representable at this grain -- see PopulationStage. // -// HOW IT IS MADE STRUCTURAL RATHER THAN CHECKED. A narrowed population is not authored -- it is -// CONSTRUCTED FROM A PARENT by filtering, and it retains the parent. Members of a stage are a subset -// of its parent's members by the meaning of filter, so a stage cannot introduce a member the parent -// lacks, and refusing a stage cannot reach back into the parent because the parent is a separate -// immutable value that the narrowing consumed rather than modified. There is no writable state for a -// regression to be written into. A lens checking this afterwards would be a second representation of -// something the construction already guarantees. +// HOW IT IS CONSTRUCTED RATHER THAN CHECKED, AND WHERE THAT STOPS. A narrowed population is not +// authored -- it is CONSTRUCTED FROM A PARENT by filtering, and it retains the parent's roster. +// Members of a stage are a subset of its parent's members by the meaning of filter, so the +// sanctioned constructor cannot introduce a member the parent lacks, and refusing a stage cannot +// reach back into the parent because the parent is a separate immutable value that the narrowing +// consumed rather than modified. +// +// That is a property of the CONSTRUCTOR, not of the type. See PopulationProvenance for the honest +// rung and its next-rung trigger: until the substrate can make a record constructor private to its +// owning module, a caller may still hand-build a ModelPopulation that violates the invariant, and +// this module makes that violation decidable rather than impossible. // The ordered stages. Each names what it is a verdict about, and BOTH are verdicts about the // RELEASE -- which is the grain this population's members are keyed at. // // WHY THE LATER STAGES ARE NOT HERE. Runtime compatibility, single- and multi-node feasibility, // context qualification and coding qualification are verdicts about a REALIZATION -- one artifact, -// one quantization, one runtime -- and this repository's own measurements falsify keying them by -// release: DeepSeek V4 Flash 0731 at IQ2_XXS is single-node feasible at a 400k floor while the SAME -// release at IQ3_S is not. A `List` cannot retain that distinction, so a stage keyed -// here would answer "DeepSeek V4 is in" for both and lose the only fact the measurement produced. -// Enumerating them anyway would be rung inflation: a stage name that reads as a modeled verdict -// while the carrier underneath cannot hold the verdict's subject. +// one quantization, one runtime -- and this repository's own measurements show two quantizations of +// ONE release reaching different verdicts at one floor. The falsifier this comment used to cite -- +// "IQ2_XXS fits at 400k and IQ3_S does not" -- was itself wrong and is withdrawn: loading the build +// on the node showed IQ3_S resident at a 400,000 window, so the memory claim reversed. What the +// measurements DO establish is the finer split: at the same 400k floor, DeepSeek V4 Flash 0731 at +// IQ2_XXS carries a retrieval receipt and is admissible, while the SAME release at UD-IQ3_S carries +// none and is UNANSWERABLE. The witness is gunbc.model.choice, exercised by +// test.claim.model.serving_choice_witness_test; the verdicts are re-derived by running it, never +// read from this sentence. +// +// A `List` cannot retain that distinction, so a stage keyed here would answer +// "DeepSeek V4 is in" for both and lose the only fact the measurement produced. Enumerating them +// anyway would be rung inflation: a stage name that reads as a modeled verdict while the carrier +// underneath cannot hold the verdict's subject. // // NEXT-RUNG TRIGGER: a RealizationIdentity distinguishing artifact x quantization x runtime, and a // RealizationPopulation whose members carry it. Those stages belong to that carrier, not this one. @@ -72,9 +84,36 @@ type ReleaseDiscoverySource { // // narrowed_from is the parent stage where one exists. Its absence marks the root, and the root is the // only population anyone authors directly. +// PROVENANCE RETAINS THE PARENT'S MEMBERSHIP, not merely its stage name. +// +// The predecessor carried `parent_stage` and a count. Neither identifies the parent: two different +// populations at the same stage are indistinguishable in it, so a "narrowing" could name a parent it +// was never derived from and nothing downstream could tell. The subset property the module claims +// was therefore unauditable at exactly the point it mattered. +// +// AN HONEST RUNG STATEMENT, because the comment above this type used to overclaim one. `narrow_population` +// does construct a subset, and no code path in this module writes a member the parent lacks. But +// `ModelPopulation` is an ordinary record: nothing in the substrate stops a caller CONSTRUCTING one +// directly with any members and any provenance it likes, including a NarrowedFrom naming a parent +// that never existed. So the class "a later stage contains a release its parent does not" sits at +// MECHANICALLY PREVENTABLE -- the sanctioned constructor cannot produce it and the retained parent +// roster makes a hand-built violation decidable -- and NOT at structurally impossible, which is what +// the original "there is no writable state for a regression to be written into" asserted. +// +// NEXT-RUNG TRIGGER: constructor privacy in the substrate -- a declared type whose values are +// producible only through named smart constructors in its owning module. With that capability +// ModelPopulation's record constructor becomes unreachable from outside, `narrow_population` and +// `open_weight_population` become the only producers, and the class reaches structural impossibility. +// Nothing short of that capability retires this row: a lens over authored constructions would be +// validation standing where construction was available, and would still miss a construction built in +// a module the lens does not read. type PopulationProvenance = DiscoveredRoot { sources: List } - | NarrowedFrom { parent_stage: PopulationStage, rejected_count: Nat } + | NarrowedFrom { + parent_stage: PopulationStage, + parent_members: List, + rejected_count: Nat, + } type ModelPopulation { stage: PopulationStage @@ -103,15 +142,55 @@ fn narrow_population( parent: ModelPopulation, stage: PopulationStage, admits: List, -) -> ModelPopulation { - let kept = filter(parent.members, r => identity_in(roster: admits, candidate: release_identity(subject: r))) - ModelPopulation { - stage: stage, - members: kept, - provenance: NarrowedFrom { - parent_stage: parent.stage, - rejected_count: length(parent.members) - length(kept), - }, +) -> NarrowingOutcome { + match population_stage_ordinal(stage: stage) > population_stage_ordinal(stage: parent.stage) { + false => NarrowingRefused { + reason: join([ + "a narrowing must move FORWARD: ", population_stage_wire(stage: parent.stage), + " -> ", population_stage_wire(stage: stage), + " is a backward or same-stage transition, which would re-key one population's members ", + "under a stage they were never adjudicated for", + ], ""), + } + true => { + let kept = filter(parent.members, r => + identity_in(roster: admits, candidate: release_identity(subject: r))) + NarrowingProduced { population: ModelPopulation { + stage: stage, + members: kept, + provenance: NarrowedFrom { + parent_stage: parent.stage, + parent_members: map(parent.members, r => release_identity(subject: r)), + rejected_count: length(parent.members) - length(kept), + }, + } } + } + } +} + +// A BACKWARD OR SAME-STAGE TRANSITION IS NOT A NARROWING. Stages are ordered, and re-labelling a +// population as an EARLIER stage would assert that failing a later test restored membership in an +// earlier one -- the exact invariant at the top of this module, inverted. +type NarrowingOutcome + = NarrowingProduced { population: ModelPopulation } + | NarrowingRefused { reason: String } + +fn population_stage_ordinal(stage: PopulationStage) -> Int { + match stage { + OpenWeightStage => 0 + LocallyPackagedStage => 1 + } +} + +// THE DECIDABLE SUBSET CHECK the retained parent roster makes possible. It is not run by +// `narrow_population` -- that constructor cannot produce a violation -- but by a consumer holding a +// population it did not build, which is the only place the question is live. +fn population_members_are_subset_of_parent(population: ModelPopulation) -> Bool { + match population.provenance { + DiscoveredRoot { sources: _ } => true + NarrowedFrom { parent_stage: _, parent_members: parents, rejected_count: _ } => + length(filter(population.members, r => + !identity_in(roster: parents, candidate: release_identity(subject: r)))) == 0 } } @@ -126,11 +205,31 @@ fn identity_in(roster: List, candidate: ReleaseIdentity) -> Boo // A completeness question is answerable only when the sources behind the population actually cover // the axis being asked about. Answering from a single distributor's catalog about the open-weight // universe is the original defect in its general form: a real count over the wrong population. +// COMPLETENESS IS ALWAYS SCOPED, and the unscoped question is the defect in its purest form. +// +// The predecessor answered `CompletenessAnswerable` for ANY PublisherReleaseCatalog source, which +// says: because someone enumerated DeepSeek's catalogue completely, this population is complete +// about the open-weight universe. It is not. A publisher catalogue is complete about ONE PUBLISHER, +// and reading it as global completeness is the original error -- a real count over the wrong +// population -- re-committed by the very function written to prevent it. +// +// So the question must name its subject. Completeness is answerable for a publisher when some +// source carries that publisher's catalogue; the universe-wide question has no answerable arm at +// all, because no source this repository can hold spans every publisher. That is not a gap waiting +// on an implementation -- it is what the sources are, said out loud. +// +// NEXT-RUNG TRIGGER: a registry-wide enumeration authority whose coverage is over publishers rather +// than over one publisher's releases. Until one exists, UniverseCompletenessUnavailable is the +// honest terminal and not a stall. +type CompletenessQuestion + = CompleteForPublisher { publisher: NonEmptyStr } + | CompleteForOpenWeightUniverse + type CompletenessVerdict - = CompletenessAnswerable { member_count: Nat } + = CompletenessAnswerable { publisher: NonEmptyStr, member_count: Nat } | CompletenessRefused { reason: String } -fn open_weight_completeness(population: ModelPopulation) -> CompletenessVerdict { +fn publisher_completeness(population: ModelPopulation, publisher: NonEmptyStr) -> CompletenessVerdict { match population.provenance { NarrowedFrom { parent_stage: p, rejected_count: _ } => CompletenessRefused { @@ -142,24 +241,54 @@ fn open_weight_completeness(population: ModelPopulation) -> CompletenessVerdict ], ""), } DiscoveredRoot { sources: sources } => - match length(filter(sources, s => coverage_spans_open_weight_universe(coverage: s.coverage))) > 0 { - true => CompletenessAnswerable { member_count: length(population.members) } + match length(filter(sources, s => + coverage_spans_publisher(coverage: s.coverage, publisher: publisher))) > 0 { + true => CompletenessAnswerable { + publisher: publisher, + member_count: length(filter(population.members, r => + (release_identity(subject: r).publisher as String) == (publisher as String))), + } false => CompletenessRefused { reason: join([ - "no discovery source claims publisher-catalog coverage; every source is a single ", - "distributor or an operator roster, so an absent release is indistinguishable from ", - "one that was never enumerated", + "no discovery source carries a release catalogue for publisher ", (publisher as String), + "; every source is a single distributor, an operator roster, or another publisher's ", + "catalogue, so an absent release is indistinguishable from one that was never enumerated", ], ""), } } } } -// A single distributor's catalog never spans the open-weight universe, and an operator roster is -// whatever a human remembered -- which is the other half of how the original census went stale. -fn coverage_spans_open_weight_universe(coverage: CoverageScope) -> Bool { +// THE UNIVERSE-WIDE QUESTION HAS NO ANSWERABLE ARM. It is a function rather than a comment so the +// refusal is executable, and it takes the population it is asked of so a caller cannot mistake it +// for a constant that might one day be edited to true. +fn open_weight_universe_completeness(population: ModelPopulation) -> CompletenessVerdict { + CompletenessRefused { + reason: join([ + "completeness about the OPEN-WEIGHT UNIVERSE is not answerable from any source shape this ", + "module can hold: a publisher catalogue is complete about one publisher, a distributor ", + "catalogue about one channel, an operator roster about what a person remembered. The ", + population_stage_wire(stage: population.stage), + " stage carries ", to_string(length(population.members)), + " member(s), which is a count over what was enumerated and not over what exists", + ], ""), + } +} + +fn completeness(population: ModelPopulation, question: CompletenessQuestion) -> CompletenessVerdict { + match question { + CompleteForPublisher { publisher: p } => + publisher_completeness(population: population, publisher: p) + CompleteForOpenWeightUniverse => open_weight_universe_completeness(population: population) + } +} + +// A publisher's catalogue spans exactly that publisher and no other. A single distributor's +// catalogue spans a channel rather than a publisher, and an operator roster is whatever a human +// remembered -- which is the other half of how the original census went stale. +fn coverage_spans_publisher(coverage: CoverageScope, publisher: NonEmptyStr) -> Bool { match coverage { - PublisherReleaseCatalog { publisher: _ } => true + PublisherReleaseCatalog { publisher: p } => (p as String) == (publisher as String) SingleDistributorCatalog { channel: _ } => false OperatorAssertedRoster => false } diff --git a/dag/test/claim/model/model_population_narrowing_witness_test.dag b/dag/test/claim/model/model_population_narrowing_witness_test.dag index f6ce1fc9512..db2b460eeff 100644 --- a/dag/test/claim/model/model_population_narrowing_witness_test.dag +++ b/dag/test/claim/model/model_population_narrowing_witness_test.dag @@ -14,11 +14,32 @@ import gunbc.model.publication { } import gunbc.model.population { ModelPopulation, - LocallyPackagedStage, + OpenWeightStage, LocallyPackagedStage, ReleaseDiscoverySource, PublisherReleaseCatalog, SingleDistributorCatalog, open_weight_population, narrow_population, - open_weight_completeness, CompletenessAnswerable, CompletenessRefused, + NarrowingOutcome, NarrowingProduced, NarrowingRefused, + population_members_are_subset_of_parent, + completeness, CompletenessQuestion, CompleteForPublisher, CompleteForOpenWeightUniverse, + CompletenessVerdict, CompletenessAnswerable, CompletenessRefused, +} + +// A narrowing that must succeed for the claim under test to mean anything. A refusal here is not a +// green: the claims below are about what a PRODUCED narrowing contains, so a refused outcome would +// satisfy them vacuously. This helper makes that distinction explicit at every call site. +fn produced_or_empty(outcome: NarrowingOutcome) -> ModelPopulation { + match outcome { + NarrowingProduced { population: p } => p + NarrowingRefused { reason: _ } => + open_weight_population(discovered: [], sources: []) + } +} + +fn was_produced(outcome: NarrowingOutcome) -> Bool { + match outcome { + NarrowingProduced { population: _ } => true + NarrowingRefused { reason: _ } => false + } } // THE FIXTURE IS THE RELEASE THAT ACTUALLY BROKE. DeepSeek V4 Flash 0731 is openly published under @@ -130,8 +151,9 @@ test fn w_ollama_cloud_presence_does_not_imply_local_weights_are_unavailable() - // invariant it was witnessing is the one asserted here, and it is the stronger statement: NO // narrowing, empty or otherwise, reaches back into its parent. test fn w_an_empty_narrowing_does_not_remove_from_the_open_weight_population() -> Bool { - let admitted_nothing = narrow_population(parent: root(), stage: LocallyPackagedStage, admits: []) - length(admitted_nothing.members) == 0 + let outcome = narrow_population(parent: root(), stage: LocallyPackagedStage, admits: []) + was_produced(outcome: outcome) + && length(produced_or_empty(outcome: outcome).members) == 0 && population_holds(population: root(), id: deepseek_v4_flash_id) } @@ -142,13 +164,16 @@ test fn w_narrowing_cannot_introduce_a_member_the_parent_lacked() -> Bool { family: "Never-Discovered" as NonEmptyStr, revision: "1" as NonEmptyStr, } - let narrowed = narrow_population( + let outcome = narrow_population( parent: root(), stage: LocallyPackagedStage, admits: [stranger, deepseek_v4_flash_id], ) - length(narrowed.members) == 1 + let narrowed = produced_or_empty(outcome: outcome) + was_produced(outcome: outcome) + && length(narrowed.members) == 1 && population_holds(population: narrowed, id: deepseek_v4_flash_id) + && population_members_are_subset_of_parent(population: narrowed) } // The positive control that the root filter discriminates at all -- without it every test above @@ -161,30 +186,74 @@ test fn w_a_closed_weight_release_is_excluded_from_the_open_weight_population() // one distributor's catalog must refuse to say how complete it is. test fn w_completeness_refuses_when_every_source_is_a_single_distributor() -> Bool { let ollama_shaped = open_weight_population(discovered: discovered(), sources: [ollama_only_source]) - match open_weight_completeness(population: ollama_shaped) { + is_refusal(verdict: completeness( + population: ollama_shaped, + question: CompleteForPublisher { publisher: "deepseek-ai" as NonEmptyStr })) +} + +fn is_refusal(verdict: CompletenessVerdict) -> Bool { + match verdict { CompletenessRefused { reason: _ } => true - CompletenessAnswerable { member_count: _ } => false + CompletenessAnswerable { publisher: _, member_count: _ } => false } } -// Positive control for RED 5: publisher-catalog coverage DOES answer. -test fn w_completeness_answers_when_a_publisher_catalog_source_is_present() -> Bool { - match open_weight_completeness(population: root()) { - CompletenessAnswerable { member_count: n } => n == 1 +// Positive control for RED 5: the publisher whose catalogue IS held does answer. +test fn w_completeness_answers_for_the_publisher_whose_catalog_is_held() -> Bool { + match completeness( + population: root(), + question: CompleteForPublisher { publisher: "deepseek-ai" as NonEmptyStr }) { + CompletenessAnswerable { publisher: p, member_count: n } => + (p as String) == "deepseek-ai" && n == 1 CompletenessRefused { reason: _ } => false } } +// RED 6 -- THE DEFECT THIS FUNCTION WAS WRITTEN TO PREVENT, RE-COMMITTED BY IT. A complete catalogue +// for deepseek-ai is complete about deepseek-ai and about nobody else. The predecessor answered +// `CompletenessAnswerable` for ANY publisher-catalogue source, so it would have reported the +// open-weight universe as fully enumerated from one publisher's index -- a real count over the wrong +// population, which is the original error verbatim. +test fn w_one_publishers_catalog_does_not_answer_for_another_publisher() -> Bool { + is_refusal(verdict: completeness( + population: root(), + question: CompleteForPublisher { publisher: "qwen" as NonEmptyStr })) +} + +// RED 7 -- and it never answers for the UNIVERSE, whatever sources are held. There is no arm that +// can return Answerable here, which is the honest terminal rather than a stall: no source shape this +// module can hold spans every publisher. +test fn w_universe_completeness_is_never_answerable() -> Bool { + is_refusal(verdict: completeness( + population: root(), question: CompleteForOpenWeightUniverse)) + && is_refusal(verdict: completeness( + population: open_weight_population(discovered: discovered(), sources: [ollama_only_source]), + question: CompleteForOpenWeightUniverse)) +} + +// RED 8 -- a BACKWARD transition is not a narrowing. Re-keying a locally-packaged population as +// open-weight would assert that failing a later test restored membership in an earlier stage, which +// is the module's invariant inverted. Same-stage is refused for the same reason. +test fn w_a_backward_or_same_stage_transition_refuses() -> Bool { + let packaged = produced_or_empty(outcome: narrow_population( + parent: root(), stage: LocallyPackagedStage, admits: [deepseek_v4_flash_id])) + !was_produced(outcome: narrow_population( + parent: packaged, stage: OpenWeightStage, admits: [deepseek_v4_flash_id])) + && !was_produced(outcome: narrow_population( + parent: packaged, stage: LocallyPackagedStage, admits: [deepseek_v4_flash_id])) + && was_produced(outcome: narrow_population( + parent: root(), stage: LocallyPackagedStage, admits: [deepseek_v4_flash_id])) +} + // A later stage refuses completeness about the universe: it is a verdict about realizations and // under-reports by construction. test fn w_a_narrowed_stage_refuses_completeness_about_the_open_weight_universe() -> Bool { - let narrowed = narrow_population( + let narrowed = produced_or_empty(outcome: narrow_population( parent: root(), stage: LocallyPackagedStage, admits: [deepseek_v4_flash_id], - ) - match open_weight_completeness(population: narrowed) { - CompletenessRefused { reason: _ } => true - CompletenessAnswerable { member_count: _ } => false - } + )) + is_refusal(verdict: completeness( + population: narrowed, + question: CompleteForPublisher { publisher: "deepseek-ai" as NonEmptyStr })) } // AGGREGATE, so the whole carrier is established in ONE corpus resolve rather than eight. It is a @@ -198,7 +267,10 @@ test fn w_all_population_claims_hold() -> Bool { && w_narrowing_cannot_introduce_a_member_the_parent_lacked() && w_a_closed_weight_release_is_excluded_from_the_open_weight_population() && w_completeness_refuses_when_every_source_is_a_single_distributor() - && w_completeness_answers_when_a_publisher_catalog_source_is_present() + && w_completeness_answers_for_the_publisher_whose_catalog_is_held() + && w_one_publishers_catalog_does_not_answer_for_another_publisher() + && w_universe_completeness_is_never_answerable() + && w_a_backward_or_same_stage_transition_refuses() && w_a_narrowed_stage_refuses_completeness_about_the_open_weight_universe() } diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index a16db8e519c..5a8278008a1 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -11,7 +11,13 @@ import gunbc.model.publication { ReleaseIdentity } import gunbc.model.choice { PrefillRateUnmeasuredAtFloor, SemanticContextUnverifiedAtFloor, FreshPrefillObservation, prefill_rate_at_floor, - OllamaRunnerMemoryObservation, runtime_memory_observation_at_floor, + ServingRealizationIdentity, serving_realization_equal, + SemanticContextEvidence, + OllamaRunnerMemoryObservation, memory_fit_evidence, + RunnerAttemptOutcome, RunnerServed, RunnerRefusedForMemory, RunnerFailedForOtherCause, + MemoryFitEvidence, MemoryFitEstablished, MemoryFitRefusedForMemory, MemoryFitContradicted, + MemoryFitAttemptFailedOtherwise, MemoryFitUnobserved, + MemoryFitContradictedAtConfiguration, MemoryAttemptFailedForNonMemoryCause, FreshSessionPrefill, WarmContinuation, serving_regime_wire, QuantizedCandidate, ServingConstraints, CandidateVerdict, ServingCandidateAdmissible, ServingCandidateRejected, @@ -19,7 +25,7 @@ import gunbc.model.choice { ServingChoice, ChoseCandidate, NoCandidateAdmissible, SelectionUnanswerable, UnanswerableCause, UnresolvedCandidateCouldWin, CrossReleaseQualityOrderAbsent, QualityRankTie, ServingCandidateUnanswerable, MissingFact, MemoryFitUnobservedAtConfiguration, - evaluate_candidate, choose_serving_candidate, + evaluate_candidate, choose_serving_candidate, verdict_is_admissible, } // THE FIXTURE IS MEASURED, not invented. Every number here was read off the running nodes or @@ -54,21 +60,54 @@ data fixture_node: NonEmptyStr = "fixture-node" as NonEmptyStr // THE READINGS, spark-a3ee, Ollama 0.32.9, one resident session, /api/ps size_vram. `size` and // `size_vram` agreed exactly on this unified-memory node, so nothing turns on which field is read. -fn iq2_xxs_footprint(depth: Nat, resident: Nat) -> OllamaRunnerMemoryObservation { +data ps_instrument: NonEmptyStr = "/api/ps size and size_vram after load at explicit num_ctx" as NonEmptyStr + +data serial_config: NonEmptyStr = "OLLAMA_NUM_PARALLEL unset; explicit num_ctx per request" as NonEmptyStr + +data ollama_runtime: NonEmptyStr = "ollama 0.32.9" as NonEmptyStr + +fn realization( + release_id: ReleaseIdentity, + quant: NonEmptyStr, + artifact: NonEmptyStr, + runtime: NonEmptyStr, + config: NonEmptyStr, +) -> ServingRealizationIdentity { + ServingRealizationIdentity { + release_id: release_id, + quant_label: quant, + artifact_ref: artifact, + runtime: runtime, + runtime_config: config, + } +} + +fn served( + r: ServingRealizationIdentity, + node: NonEmptyStr, + instrument: NonEmptyStr, + depth: Nat, + sessions: Nat, + resident: Nat, +) -> OllamaRunnerMemoryObservation { OllamaRunnerMemoryObservation { - realization: "hf.co/antirez/deepseek-v4-gguf:latest" as NonEmptyStr, - runtime: "ollama 0.32.9" as NonEmptyStr, - runtime_config: "OLLAMA_NUM_PARALLEL unset; explicit num_ctx per request" as NonEmptyStr, - node: "spark-a3ee" as NonEmptyStr, - instrument: "/api/ps size and size_vram after load at explicit num_ctx" as NonEmptyStr, + realization: r, + node: node, + instrument: instrument, context_depth: token_count(count: depth), - concurrent_sessions: 1, - reported_total_buffer: byte_size(count: resident), - reported_gpu_buffer: byte_size(count: resident), - generation_succeeded: true, + concurrent_sessions: sessions, + outcome: RunnerServed { + reported_total_buffer: byte_size(count: resident), + reported_gpu_buffer: byte_size(count: resident), + }, } } +fn iq2_xxs_footprint(depth: Nat, resident: Nat) -> OllamaRunnerMemoryObservation { + served(r: r_iq2_xxs, node: serving_node, instrument: ps_instrument, + depth: depth, sessions: 1, resident: resident) +} + // RELEASE IDENTITIES, exact rather than display names. Publisher, family and revision together are // what decide whether two candidates share an ordinal quality scale, so the selector is keyed by // this and never by a name string: two revisions of one family share a name and would merge under a @@ -91,19 +130,72 @@ data qwen_release: ReleaseIdentity = ReleaseIdentity { revision: "35B" as NonEmptyStr, } +data r_iq2_xxs: ServingRealizationIdentity = realization( + release_id: deepseek_v4_flash_release, + quant: "IQ2_XXS" as NonEmptyStr, + artifact: "hf.co/antirez/deepseek-v4-gguf:latest" as NonEmptyStr, + runtime: ollama_runtime, + config: serial_config, +) + +data r_iq3_s: ServingRealizationIdentity = realization( + release_id: deepseek_v4_flash_release, + quant: "UD-IQ3_S" as NonEmptyStr, + artifact: "deepseek-v4-flash:iq3s" as NonEmptyStr, + runtime: ollama_runtime, + config: serial_config, +) + +data r_unmeasured: ServingRealizationIdentity = realization( + release_id: deepseek_v4_flash_release, + quant: "UD-IQ4_XS" as NonEmptyStr, + artifact: "deepseek-v4-flash:iq4xs" as NonEmptyStr, + runtime: ollama_runtime, + config: serial_config, +) + +data r_qwen: ServingRealizationIdentity = realization( + release_id: qwen_release, + quant: "Q8_0" as NonEmptyStr, + artifact: "qwen3.6:35b" as NonEmptyStr, + runtime: ollama_runtime, + config: serial_config, +) + +fn fixture_realization(quant: NonEmptyStr) -> ServingRealizationIdentity { + realization( + release_id: fixture_release, + quant: quant, + artifact: "declared fixture, not an artifact" as NonEmptyStr, + runtime: "fixture-runtime" as NonEmptyStr, + config: "declared fixture, not a configuration" as NonEmptyStr, + ) +} + +data r_fixture_low: ServingRealizationIdentity = fixture_realization(quant: "LOW" as NonEmptyStr) +data r_fixture_high: ServingRealizationIdentity = fixture_realization(quant: "HIGH" as NonEmptyStr) +data r_tied_a: ServingRealizationIdentity = fixture_realization(quant: "TIED-A" as NonEmptyStr) +data r_tied_b: ServingRealizationIdentity = fixture_realization(quant: "TIED-B" as NonEmptyStr) + +fn verified_at(r: ServingRealizationIdentity, depth: Nat) -> SemanticContextEvidence? { + Present { value: SemanticContextEvidence { + realization: r, + verified_to: token_count(count: depth), + } } +} + // The two builds actually installed. Quality rank is ordinal WITHIN this identity: 3-bit outranks // 2-bit. It says nothing about any other model, which is why cross-release quality never enters // this function as a comparable number. data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { - identity: deepseek_v4_flash_release, - quant_label: "IQ2_XXS" as NonEmptyStr, + realization: r_iq2_xxs, quality_rank: 2, declared_context: token_count(count: 1048576), - semantic_context_verified_to: Present { value: token_count(count: 400060) }, + semantic_context: verified_at(r: r_iq2_xxs, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { depth: token_count(count: 160060), rate: tokens_per_second(count: 253) }, - FreshPrefillObservation { depth: token_count(count: 255061), rate: tokens_per_second(count: 197) }, - FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, + FreshPrefillObservation { realization: r_iq2_xxs, depth: token_count(count: 160060), rate: tokens_per_second(count: 253) }, + FreshPrefillObservation { realization: r_iq2_xxs, depth: token_count(count: 255061), rate: tokens_per_second(count: 197) }, + FreshPrefillObservation { realization: r_iq2_xxs, depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, ], runtime_memory_observations: [ iq2_xxs_footprint(depth: 131072, resident: 86532465622), @@ -114,25 +206,14 @@ data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { } data build_iq3_s: QuantizedCandidate = QuantizedCandidate { - identity: deepseek_v4_flash_release, - quant_label: "UD-IQ3_S" as NonEmptyStr, + realization: r_iq3_s, quality_rank: 3, declared_context: token_count(count: 1048576), - semantic_context_verified_to: none, + semantic_context: none, fresh_prefill_observations: [], runtime_memory_observations: [ - OllamaRunnerMemoryObservation { - realization: "deepseek-v4-flash:iq3s" as NonEmptyStr, - runtime: "ollama 0.32.9" as NonEmptyStr, - runtime_config: "OLLAMA_NUM_PARALLEL unset; explicit num_ctx per request" as NonEmptyStr, - node: "spark-a3ee" as NonEmptyStr, - instrument: "/api/ps size and size_vram after load at explicit num_ctx" as NonEmptyStr, - context_depth: token_count(count: 400000), - concurrent_sessions: 1, - reported_total_buffer: byte_size(count: 116970000000), - reported_gpu_buffer: byte_size(count: 116970000000), - generation_succeeded: true, - }, + served(r: r_iq3_s, node: serving_node, instrument: ps_instrument, + depth: 400000, sessions: 1, resident: 116970000000), ], } @@ -162,7 +243,7 @@ fn chosen_label(choice: ServingChoice) -> String { NoCandidateAdmissible { rejections: _ } => "none" ChoseCandidate { verdict: v } => match v { - ServingCandidateAdmissible { candidate: c, fit: _ } => c.quant_label as String + ServingCandidateAdmissible { candidate: c, fit: _ } => c.realization.quant_label as String ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => "none" ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => "none" } @@ -200,6 +281,8 @@ test fn w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolve SemanticContextUnverifiedAtFloor { floor: f, declared: d } => token_count_value(t: f) == 400000 && token_count_value(t: d) == 1048576 MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false + MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false + MemoryAttemptFailedForNonMemoryCause { cause: _ } => false PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false } ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false @@ -213,28 +296,51 @@ test fn w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolve test fn w_the_real_roster_refuses_while_the_higher_rank_is_unresolved() -> Bool { match choose_serving_candidate( candidates: installed(), node: serving_node, constraints: floor_400k) { - SelectionUnanswerable { cause: _, evaluations: e } => length(e) == 2 + SelectionUnanswerable { cause: c, evaluations: e } => + length(e) == 2 + && match c { + UnresolvedCandidateCouldWin { unresolved_count: n } => n == 1 + CrossReleaseQualityOrderAbsent { distinct_release_count: _ } => false + QualityRankTie { tied_count: _, rank: _ } => false + } + && length(filter(e, v => verdict_is_admissible(verdict: v))) == 1 + && length(filter(e, v => verdict_is_unanswerable(verdict: v))) == 1 ChoseCandidate { verdict: _ } => false NoCandidateAdmissible { rejections: _ } => false } } +fn verdict_is_unanswerable(verdict: CandidateVerdict) -> Bool { + match verdict { + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => true + ServingCandidateAdmissible { candidate: _, fit: _ } => false + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false + } +} + // THE POSITIVE CONTROL FOR THE FOOTPRINT AXIS: the 2-bit build IS resident-established at the floor, // so the refusal above is not a footprint carrier that answers nothing. Its qualifying reading is // the 400,000 one rather than the 1,048,576 one -- the tightest sound bound at or beyond the floor. test fn w_the_two_bit_build_is_resident_established_at_the_floor() -> Bool { - match runtime_memory_observation_at_floor( + match memory_fit_evidence( observations: build_iq2_xxs.runtime_memory_observations, + realization: r_iq2_xxs, floor: token_count(count: 400000), hot_sessions: 1, node: serving_node, ) { - Present { value: o } => - byte_size_count(b: o.reported_total_buffer) == 87692389907 - && byte_size_count(b: o.reported_gpu_buffer) == 87692389907 - && token_count_value(t: o.context_depth) == 400000 - && o.generation_succeeded - Absent => false + MemoryFitEstablished { by: o } => + token_count_value(t: o.context_depth) == 400000 + && match o.outcome { + RunnerServed { reported_total_buffer: t, reported_gpu_buffer: g } => + byte_size_count(b: t) == 87692389907 && byte_size_count(b: g) == 87692389907 + RunnerRefusedForMemory { detail: _ } => false + RunnerFailedForOtherCause { cause: _ } => false + } + MemoryFitRefusedForMemory { by: _ } => false + MemoryFitContradicted { served: _, refused: _ } => false + MemoryFitAttemptFailedOtherwise { by: _ } => false + MemoryFitUnobserved => false } } @@ -243,15 +349,103 @@ test fn w_the_two_bit_build_is_resident_established_at_the_floor() -> Bool { // that positively answers one hot session at 400k above. test fn w_a_shallow_or_more_concurrent_demand_has_no_qualifying_footprint() -> Bool { let obs = build_iq2_xxs.runtime_memory_observations - match runtime_memory_observation_at_floor( - observations: obs, floor: token_count(count: 2000000), hot_sessions: 1, node: serving_node) { + is_unobserved(evidence: memory_fit_evidence( + observations: obs, realization: r_iq2_xxs, + floor: token_count(count: 2000000), hot_sessions: 1, node: serving_node)) + && is_unobserved(evidence: memory_fit_evidence( + observations: obs, realization: r_iq2_xxs, + floor: token_count(count: 400000), hot_sessions: 2, node: serving_node)) +} + +fn is_unobserved(evidence: MemoryFitEvidence) -> Bool { + match evidence { + MemoryFitUnobserved => true + MemoryFitEstablished { by: _ } => false + MemoryFitRefusedForMemory { by: _ } => false + MemoryFitContradicted { served: _, refused: _ } => false + MemoryFitAttemptFailedOtherwise { by: _ } => false + } +} + +// ================= THE NEGATIVE CONTROLS FOR THE IDENTITY JOIN ================= +// +// Provenance fields that are carried but never matched are decoration. Each control below takes the +// SAME observation list that positively answers above, asks it on behalf of a realization differing +// in exactly ONE component, and requires MemoryFitUnobserved. Together with the positive control +// they establish that every component of the join key is load-bearing: flip any one and the evidence +// stops answering. +fn iq2_xxs_answers_for(r: ServingRealizationIdentity, node: NonEmptyStr) -> Bool { + match memory_fit_evidence( + observations: build_iq2_xxs.runtime_memory_observations, + realization: r, + floor: token_count(count: 400000), + hot_sessions: 1, + node: node, + ) { + MemoryFitEstablished { by: _ } => true + MemoryFitRefusedForMemory { by: _ } => false + MemoryFitContradicted { served: _, refused: _ } => false + MemoryFitAttemptFailedOtherwise { by: _ } => false + MemoryFitUnobserved => false + } +} + +test fn w_evidence_from_another_realization_or_node_cannot_qualify() -> Bool { + let wrong_release = realization( + release_id: qwen_release, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_ref, + runtime: r_iq2_xxs.runtime, config: r_iq2_xxs.runtime_config) + let wrong_quant = realization( + release_id: r_iq2_xxs.release_id, quant: "IQ3_S" as NonEmptyStr, artifact: r_iq2_xxs.artifact_ref, + runtime: r_iq2_xxs.runtime, config: r_iq2_xxs.runtime_config) + let wrong_artifact = realization( + release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, + artifact: "hf.co/someone-else/deepseek-v4-gguf:latest" as NonEmptyStr, + runtime: r_iq2_xxs.runtime, config: r_iq2_xxs.runtime_config) + let wrong_runtime = realization( + release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_ref, + runtime: "ollama 0.33.0" as NonEmptyStr, config: r_iq2_xxs.runtime_config) + let wrong_config = realization( + release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_ref, + runtime: r_iq2_xxs.runtime, config: "OLLAMA_NUM_PARALLEL=4" as NonEmptyStr) + iq2_xxs_answers_for(r: r_iq2_xxs, node: serving_node) + && !iq2_xxs_answers_for(r: wrong_release, node: serving_node) + && !iq2_xxs_answers_for(r: wrong_quant, node: serving_node) + && !iq2_xxs_answers_for(r: wrong_artifact, node: serving_node) + && !iq2_xxs_answers_for(r: wrong_runtime, node: serving_node) + && !iq2_xxs_answers_for(r: wrong_config, node: serving_node) + && !iq2_xxs_answers_for(r: r_iq2_xxs, node: fixture_node) +} + +// A RETRIEVAL RECEIPT DOES NOT TRANSFER between realizations. The candidate below is the installed +// IQ3_S -- which has no receipt of its own -- handed IQ2_XXS's receipt. Before the join it would +// have been admitted on borrowed evidence, because the receipt was a bare depth with no subject. +test fn w_a_retrieval_receipt_from_another_realization_does_not_qualify() -> Bool { + let borrowed = QuantizedCandidate { + realization: r_iq3_s, + quality_rank: build_iq3_s.quality_rank, + declared_context: build_iq3_s.declared_context, + semantic_context: verified_at(r: r_iq2_xxs, depth: 400060), + fresh_prefill_observations: build_iq3_s.fresh_prefill_observations, + runtime_memory_observations: build_iq3_s.runtime_memory_observations, + } + match evaluate_candidate(candidate: borrowed, node: serving_node, constraints: floor_400k) { + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => + is_semantic_gap(missing: m) + ServingCandidateAdmissible { candidate: _, fit: _ } => false + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false + } +} + +// A PREFILL RECEIPT DOES NOT TRANSFER EITHER, on the same reasoning and the same axis. +test fn w_a_prefill_receipt_from_another_realization_does_not_qualify() -> Bool { + match prefill_rate_at_floor( + observations: build_iq2_xxs.fresh_prefill_observations, + realization: r_qwen, + floor: token_count(count: 400000), + regime: FreshSessionPrefill, + ) { Present { value: _ } => false - Absent => - match runtime_memory_observation_at_floor( - observations: obs, floor: token_count(count: 400000), hot_sessions: 2, node: serving_node) { - Present { value: _ } => false - Absent => true - } + Absent => true } } @@ -266,22 +460,24 @@ test fn w_a_shallow_or_more_concurrent_demand_has_no_qualifying_footprint() -> B // SYNTHETIC footprints for the declared ordering fixtures below. These are not fleet readings and // say so in every field: the node is named "fixture", so a real constraint can never be answered by // one, and the ordering claims they support are about the SELECTOR and not about any hardware. -fn fixture_footprint(resident: Nat) -> OllamaRunnerMemoryObservation { - fixture_load(resident: resident, succeeded: true) +data fixture_instrument: NonEmptyStr = "declared fixture, not a reading" as NonEmptyStr + +fn fixture_footprint(r: ServingRealizationIdentity, resident: Nat) -> OllamaRunnerMemoryObservation { + served(r: r, node: fixture_node, instrument: fixture_instrument, + depth: 1048576, sessions: 1, resident: resident) } -fn fixture_load(resident: Nat, succeeded: Bool) -> OllamaRunnerMemoryObservation { +fn fixture_attempt( + r: ServingRealizationIdentity, + outcome: RunnerAttemptOutcome, +) -> OllamaRunnerMemoryObservation { OllamaRunnerMemoryObservation { - realization: "fixture" as NonEmptyStr, - runtime: "fixture" as NonEmptyStr, - runtime_config: "fixture" as NonEmptyStr, - node: "fixture-node" as NonEmptyStr, - instrument: "declared fixture, not a reading" as NonEmptyStr, + realization: r, + node: fixture_node, + instrument: fixture_instrument, context_depth: token_count(count: 1048576), concurrent_sessions: 1, - reported_total_buffer: byte_size(count: resident), - reported_gpu_buffer: byte_size(count: resident), - generation_succeeded: succeeded, + outcome: outcome, } } @@ -289,27 +485,25 @@ fn fixture_load(resident: Nat, succeeded: Bool) -> OllamaRunnerMemoryObservation // release, so a maximum over them exists, and they differ only in quality_rank and weights. Without // this control every other claim here is satisfied by a function that returns its first argument. data fixture_low_rank: QuantizedCandidate = QuantizedCandidate { - identity: fixture_release, - quant_label: "LOW" as NonEmptyStr, + realization: r_fixture_low, quality_rank: 1, declared_context: token_count(count: 1048576), - semantic_context_verified_to: Present { value: token_count(count: 400060) }, + semantic_context: verified_at(r: r_fixture_low, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + FreshPrefillObservation { realization: r_fixture_low, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], - runtime_memory_observations: [fixture_footprint(resident: 28000000000)], + runtime_memory_observations: [fixture_footprint(r: r_fixture_low, resident: 28000000000)], } data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { - identity: fixture_release, - quant_label: "HIGH" as NonEmptyStr, + realization: r_fixture_high, quality_rank: 9, declared_context: token_count(count: 1048576), - semantic_context_verified_to: Present { value: token_count(count: 400060) }, + semantic_context: verified_at(r: r_fixture_high, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 400) }, + FreshPrefillObservation { realization: r_fixture_high, depth: token_count(count: 400060), rate: tokens_per_second(count: 400) }, ], - runtime_memory_observations: [fixture_footprint(resident: 38000000000)], + runtime_memory_observations: [fixture_footprint(r: r_fixture_high, resident: 38000000000)], } test fn w_the_higher_quality_candidate_wins_when_both_are_admissible() -> Bool { @@ -332,6 +526,8 @@ test fn w_a_declared_context_never_qualifies_without_a_retrieval_receipt() -> Bo SemanticContextUnverifiedAtFloor { floor: _, declared: d } => token_count_value(t: d) == 1048576 MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false + MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false + MemoryAttemptFailedForNonMemoryCause { cause: _ } => false PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false } ServingCandidateAdmissible { candidate: _, fit: _ } => false @@ -344,9 +540,11 @@ test fn w_a_declared_context_never_qualifies_without_a_retrieval_receipt() -> Bo test fn w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() -> Bool { match prefill_rate_at_floor( observations: [FreshPrefillObservation { + realization: r_iq2_xxs, depth: token_count(count: 160060), rate: tokens_per_second(count: 253), }], + realization: r_iq2_xxs, floor: token_count(count: 400000), regime: FreshSessionPrefill, ) { @@ -369,26 +567,8 @@ test fn w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() -> Bool // that was attempted on the node and DID NOT PRODUCE A TOKEN. That is an observation, not a // calculation, and it is the arm a real out-of-memory load would populate. test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { - let did_not_load_low = QuantizedCandidate { - identity: fixture_low_rank.identity, - quant_label: fixture_low_rank.quant_label, - quality_rank: fixture_low_rank.quality_rank, - declared_context: fixture_low_rank.declared_context, - semantic_context_verified_to: fixture_low_rank.semantic_context_verified_to, - fresh_prefill_observations: fixture_low_rank.fresh_prefill_observations, - runtime_memory_observations: [fixture_load(resident: 28000000000, succeeded: false)], - } - let did_not_load_high = QuantizedCandidate { - identity: fixture_high_rank.identity, - quant_label: fixture_high_rank.quant_label, - quality_rank: fixture_high_rank.quality_rank, - declared_context: fixture_high_rank.declared_context, - semantic_context_verified_to: fixture_high_rank.semantic_context_verified_to, - fresh_prefill_observations: fixture_high_rank.fresh_prefill_observations, - runtime_memory_observations: [fixture_load(resident: 38000000000, succeeded: false)], - } match choose_serving_candidate( - candidates: [did_not_load_low, did_not_load_high], + candidates: [oom_low(), oom_high()], node: fixture_node, constraints: floor_400k, ) { @@ -414,16 +594,19 @@ test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { // non-fresh arms of prefill_rate_at_floor have nothing to read. test fn w_a_fresh_prefill_receipt_does_not_answer_a_warm_continuation_floor() -> Bool { let deep_fresh = [FreshPrefillObservation { + realization: r_iq2_xxs, depth: token_count(count: 400060), rate: tokens_per_second(count: 135), }] match prefill_rate_at_floor( - observations: deep_fresh, floor: token_count(count: 400000), regime: WarmContinuation, + observations: deep_fresh, realization: r_iq2_xxs, + floor: token_count(count: 400000), regime: WarmContinuation, ) { Present { value: _ } => false Absent => match prefill_rate_at_floor( - observations: deep_fresh, floor: token_count(count: 400000), regime: FreshSessionPrefill, + observations: deep_fresh, realization: r_iq2_xxs, + floor: token_count(count: 400000), regime: FreshSessionPrefill, ) { Absent => false Present { value: r } => tokens_per_second_count(r: r) == 135 @@ -446,6 +629,8 @@ test fn w_a_warm_floor_over_fresh_only_receipts_is_unanswerable() -> Bool { PrefillRateUnmeasuredAtFloor { floor: _, regime: g } => serving_regime_wire(regime: g) == "warm-continuation" MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false + MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false + MemoryAttemptFailedForNonMemoryCause { cause: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false } ServingCandidateAdmissible { candidate: _, fit: _ } => false @@ -453,6 +638,132 @@ test fn w_a_warm_floor_over_fresh_only_receipts_is_unanswerable() -> Bool { } } +fn is_semantic_gap(missing: MissingFact) -> Bool { + match missing { + SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => true + MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false + MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false + MemoryAttemptFailedForNonMemoryCause { cause: _ } => false + PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false + } +} + +fn with_memory_observations( + base: QuantizedCandidate, + observations: List, +) -> QuantizedCandidate { + QuantizedCandidate { + realization: base.realization, + quality_rank: base.quality_rank, + declared_context: base.declared_context, + semantic_context: base.semantic_context, + fresh_prefill_observations: base.fresh_prefill_observations, + runtime_memory_observations: observations, + } +} + +data out_of_memory: RunnerAttemptOutcome = RunnerRefusedForMemory { + detail: "llama_model_load: unable to allocate KV cache buffer" as NonEmptyStr, +} + +fn oom_low() -> QuantizedCandidate { + with_memory_observations(base: fixture_low_rank, + observations: [fixture_attempt(r: r_fixture_low, outcome: out_of_memory)]) +} + +fn oom_high() -> QuantizedCandidate { + with_memory_observations(base: fixture_high_rank, + observations: [fixture_attempt(r: r_fixture_high, outcome: out_of_memory)]) +} + +// ================= WHAT AN UNSUCCESSFUL ATTEMPT MEANS ================= +// +// A Bool `generation_succeeded` mapped every failure to a memory verdict. It does not follow: a +// template error, a cancelled request or a transport fault produces no token and says nothing about +// whether the weights and cache fit. Only the typed insufficient-memory refusal may reject; the +// other failure is UNANSWERABLE and names its cause. +test fn w_only_a_typed_memory_refusal_rejects_on_memory() -> Bool { + let broken = with_memory_observations(base: fixture_low_rank, observations: [ + fixture_attempt(r: r_fixture_low, outcome: RunnerFailedForOtherCause { + cause: "chat template rendering failed" as NonEmptyStr, + }), + ]) + let rejects_on_memory = match evaluate_candidate( + candidate: oom_low(), node: fixture_node, constraints: floor_8k) { + ServingCandidateRejected { identity: _, quant_label: _, axis: a } => + match a { + DoesNotFitMemory { observed: _ } => true + ContextBelowFloor { declared: _, floor: _ } => false + PrefillBelowFloor { measured: _, floor: _ } => false + } + ServingCandidateAdmissible { candidate: _, fit: _ } => false + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => false + } + let other_is_unanswerable = match evaluate_candidate( + candidate: broken, node: fixture_node, constraints: floor_8k) { + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => + match m { + MemoryAttemptFailedForNonMemoryCause { cause: c } => + (c as String) == "chat template rendering failed" + MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false + MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false + SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false + PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false + } + ServingCandidateAdmissible { candidate: _, fit: _ } => false + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false + } + rejects_on_memory && other_is_unanswerable +} + +// ================= MAGNITUDE DOES NOT DECIDE BETWEEN CONTRADICTIONS ================= +// +// The fold this replaced picked the SMALLEST reported buffer and only then read the success flag, so +// a small failed receipt beat a large successful one and flipped admission into rejection -- and the +// reverse pairing admitted a candidate whose attempt had failed. Both orders are exercised here, and +// both must refuse: two qualifying attempts that disagree are a question, not a measurement. +fn is_contradiction_refusal(verdict: CandidateVerdict) -> Bool { + match verdict { + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => + match m { + MemoryFitContradictedAtConfiguration { floor: f, sessions: n } => + token_count_value(t: f) == 8192 && n == 1 + MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false + MemoryAttemptFailedForNonMemoryCause { cause: _ } => false + SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false + PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false + } + ServingCandidateAdmissible { candidate: _, fit: _ } => false + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false + } +} + +test fn w_contradictory_attempts_refuse_in_both_orders_and_at_both_magnitudes() -> Bool { + let big_success = fixture_footprint(r: r_fixture_low, resident: 99000000000) + let small_failure = fixture_attempt(r: r_fixture_low, outcome: out_of_memory) + is_contradiction_refusal(verdict: evaluate_candidate( + candidate: with_memory_observations(base: fixture_low_rank, + observations: [big_success, small_failure]), + node: fixture_node, constraints: floor_8k)) + && is_contradiction_refusal(verdict: evaluate_candidate( + candidate: with_memory_observations(base: fixture_low_rank, + observations: [small_failure, big_success]), + node: fixture_node, constraints: floor_8k)) +} + +// And the POSITIVE CONTROL for the same pair: with the memory refusal removed, the identical +// large-buffer success admits. So the refusal above is the disagreement and not the magnitude. +test fn w_the_large_buffer_success_alone_still_admits() -> Bool { + match evaluate_candidate( + candidate: with_memory_observations(base: fixture_low_rank, + observations: [fixture_footprint(r: r_fixture_low, resident: 99000000000)]), + node: fixture_node, constraints: floor_8k) { + ServingCandidateAdmissible { candidate: _, fit: _ } => true + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => false + } +} + test fn w_all_serving_choice_claims_hold() -> Bool { w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolved() && w_the_real_roster_refuses_while_the_higher_rank_is_unresolved() @@ -465,6 +776,12 @@ test fn w_all_serving_choice_claims_hold() -> Bool { && w_nothing_admissible_refuses_and_reports_every_rejection() && w_a_fresh_prefill_receipt_does_not_answer_a_warm_continuation_floor() && w_a_warm_floor_over_fresh_only_receipts_is_unanswerable() + && w_evidence_from_another_realization_or_node_cannot_qualify() + && w_a_retrieval_receipt_from_another_realization_does_not_qualify() + && w_a_prefill_receipt_from_another_realization_does_not_qualify() + && w_only_a_typed_memory_refusal_rejects_on_memory() + && w_contradictory_attempts_refuse_in_both_orders_and_at_both_magnitudes() + && w_the_large_buffer_success_alone_still_admits() } // ======================= THE ANSWERABILITY CLAIMS ======================= @@ -474,13 +791,12 @@ test fn w_all_serving_choice_claims_hold() -> Bool { // A candidate whose KV cost was never measured has not FAILED the memory test -- it has not taken // it. The selector must refuse rather than hand the win to the measured candidate by default. data build_unmeasured: QuantizedCandidate = QuantizedCandidate { - identity: deepseek_v4_flash_release, - quant_label: "UD-IQ4_XS" as NonEmptyStr, + realization: r_unmeasured, quality_rank: 4, declared_context: token_count(count: 1048576), - semantic_context_verified_to: Present { value: token_count(count: 400060) }, + semantic_context: verified_at(r: r_unmeasured, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, + FreshPrefillObservation { realization: r_unmeasured, depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, ], runtime_memory_observations: [], } @@ -491,6 +807,8 @@ test fn w_an_unmeasured_candidate_is_unanswerable_not_rejected() -> Bool { match m { MemoryFitUnobservedAtConfiguration { floor: f, sessions: n } => token_count_value(t: f) == 400000 && n == 1 + MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false + MemoryAttemptFailedForNonMemoryCause { cause: _ } => false PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false } @@ -518,27 +836,16 @@ test fn w_selection_refuses_while_an_unresolved_candidate_could_win() -> Bool { // Preference is ordinal WITHIN a release, so two admissible releases have no join and a maximum over // them does not exist. Returning one would manufacture an ordering the inputs never contained. data other_release: QuantizedCandidate = QuantizedCandidate { - identity: qwen_release, - quant_label: "Q8_0" as NonEmptyStr, + realization: r_qwen, quality_rank: 3, declared_context: token_count(count: 1048576), - semantic_context_verified_to: Present { value: token_count(count: 400060) }, + semantic_context: verified_at(r: r_qwen, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 900) }, + FreshPrefillObservation { realization: r_qwen, depth: token_count(count: 400060), rate: tokens_per_second(count: 900) }, ], runtime_memory_observations: [ - OllamaRunnerMemoryObservation { - realization: "qwen3.6:35b" as NonEmptyStr, - runtime: "ollama 0.32.9" as NonEmptyStr, - runtime_config: "declared fixture, not a reading" as NonEmptyStr, - node: "spark-a3ee" as NonEmptyStr, - instrument: "declared fixture, not a reading" as NonEmptyStr, - context_depth: token_count(count: 1048576), - concurrent_sessions: 1, - reported_total_buffer: byte_size(count: 30600000000), - reported_gpu_buffer: byte_size(count: 30600000000), - generation_succeeded: true, - }, + served(r: r_qwen, node: serving_node, instrument: fixture_instrument, + depth: 1048576, sessions: 1, resident: 30600000000), ], } @@ -580,27 +887,25 @@ test fn w_all_answerability_claims_hold() -> Bool { // would return a different winner for the same inputs, which is why "stable under re-ordering" and // "keep the earlier one" cannot both be true. data fixture_tied_a: QuantizedCandidate = QuantizedCandidate { - identity: fixture_release, - quant_label: "TIED-A" as NonEmptyStr, + realization: r_tied_a, quality_rank: 5, declared_context: token_count(count: 1048576), - semantic_context_verified_to: Present { value: token_count(count: 400060) }, + semantic_context: verified_at(r: r_tied_a, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + FreshPrefillObservation { realization: r_tied_a, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], - runtime_memory_observations: [fixture_footprint(resident: 28000000000)], + runtime_memory_observations: [fixture_footprint(r: r_tied_a, resident: 28000000000)], } data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { - identity: fixture_release, - quant_label: "TIED-B" as NonEmptyStr, + realization: r_tied_b, quality_rank: 5, declared_context: token_count(count: 1048576), - semantic_context_verified_to: Present { value: token_count(count: 400060) }, + semantic_context: verified_at(r: r_tied_b, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + FreshPrefillObservation { realization: r_tied_b, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], - runtime_memory_observations: [fixture_footprint(resident: 29000000000)], + runtime_memory_observations: [fixture_footprint(r: r_tied_b, resident: 29000000000)], } fn is_tie_refusal(choice: ServingChoice) -> Bool { From a239fe39ebcad834e6618720c31fec9a705475f5 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 22:55:38 +0000 Subject: [PATCH 15/22] Content-address the identity, refuse to generalize a failure downward, seal the population The three blockers from the side chat's re-review of b0ca779d1df. (1) EXACT IDENTITY. artifact, runtime and runtime-config were freely authored strings, and a string is a claim about an identity rather than the identity. The artifact axis is now a ContentHash -- Ollama reports a digest for every loaded model, and a tag is a MOVING reference, so evidence keyed by tag silently spans two artifacts. The configuration axis is a digest of the RESOLVED ARGV rather than the desired environment: this fleet ran for hours with a desired context value never rendered onto either host, and the runtime resolves flags of its own that no environment variable names. The runtime axis is sealed with one mint, because a runtime build has no digest this repository observes yet; that axis alone carries the trigger. The whole identity is sole_constructor with a single mint, and cross-family digest pairs are refused rather than collapsed to false. Fresh-prefill evidence now binds the host and the concurrency condition too. A rate measured on an idle host is not the rate a caller meets behind three other prefills -- this fleet's own 300x latency scare was exactly that, a queue read as a property of the model. (2) FAILURE DOES NOT GENERALIZE DOWNWARD. One qualifying filter served both outcomes, so an out-of-memory refusal at 1,048,576 was read as a refusal of every shallower demand -- rejecting configurations that had never been tested. A success at a harder point still establishes an easier one, because the deeper load asks strictly more of the same machine; a memory refusal now qualifies at its EXACT point and nowhere else. Witnesses cover both axes, each with the positive control that keeps the refusal from being satisfied by a filter that never matches. (3) THE POPULATION IS SEALED. The previous note claimed mechanical prevention and named constructor privacy as an unavailable capability. It was already in the tree: sole_constructor, used by extdeps.pin, extdeps.exec.command and three others for this exact shape. So the row named a trigger that existed and would have sat below its ceiling forever while reading as an honest stall. ModelPopulation and PopulationProvenance are now sole_constructor with two mints, the class is structurally impossible within this authority, and the refusal is measured by execution at a fixture boundary with the enrolled cross-module positive control named. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01U397y4s3dSBof7vGPAX87G --- dag/gunbc/model/choice.dag | 174 +++++++++++- dag/gunbc/model/population.dag | 76 ++++-- .../model/serving_choice_witness_test.dag | 251 +++++++++++++++--- 3 files changed, 417 insertions(+), 84 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index 9cd9362d7f4..9d8406ca170 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -3,6 +3,10 @@ module gunbc.model.choice import std.types { String, Bool, List, NonEmptyStr, Int } import std.nat { Nat } import gunbc.model.publication { ReleaseIdentity, release_identity_equal } +import std.content_hash { + ContentHash, ContentHashComparison, ContentHashEqual, ContentHashDifferent, + ContentHashCrossFamilyIncomparable, compare_content_hash, +} import std.measure { ByteSize, byte_size, byte_size_count, TokenCount, token_count, token_count_value, @@ -74,20 +78,90 @@ import std.measure { // retrieval evidence from B, prefill evidence from C, admitted as one candidate that never existed. // So every evidence axis joins THIS identity, and an observation that does not carry it cannot // answer for this candidate at all. -type ServingRealizationIdentity { +// THE THREE AXES BELOW WERE FREELY AUTHORED STRINGS, and a string is a claim about an identity +// rather than the identity. Two of them can be content-addressed from facts the fleet already +// emits, and the third is sealed so the module that owns the concept is the only mint. +// +// ARTIFACT: content-addressed. Ollama reports a digest for every loaded model on /api/ps, so the +// artifact axis carries that digest and not a tag. A tag is a MOVING reference -- 'latest' resolves +// to different bytes on different days -- so evidence keyed by tag silently spans two artifacts, +// which is precisely the failure the join was introduced to close, one level down. +// +// RUNTIME CONFIGURATION: content-addressed over the RESOLVED ARGV, not the desired environment. The +// distinction is measured, not theoretical: this fleet ran for hours with a desired context value +// that had never been rendered onto either host, and the runtime also resolves flags of its own +// (--flash, --load-mode) that no environment variable names. The argv is what the runner actually +// received, so hashing it is the only key under which two readings are of the same configuration. +// +// RUNTIME: sealed rather than hashed, because a runtime build has no digest this repository +// observes today. It is a sole_constructor carrier with one mint, so a bare string cannot travel as +// a runtime identity. NEXT-RUNG TRIGGER for this axis alone: an observed runtime build identity -- +// the binary's own digest or a version endpoint reading -- at which point it joins the two above. +type ServingRuntimeIdentity sole_constructor { + name: NonEmptyStr + version: NonEmptyStr +} + +fn serving_runtime_identity(name: NonEmptyStr, version: NonEmptyStr) -> ServingRuntimeIdentity { + ServingRuntimeIdentity { name: name, version: version } +} + +fn serving_runtime_equal(a: ServingRuntimeIdentity, b: ServingRuntimeIdentity) -> Bool { + (a.name as String) == (b.name as String) && (a.version as String) == (b.version as String) +} + +type ResolvedRuntimeConfiguration sole_constructor { + argv_digest: ContentHash +} + +fn resolved_runtime_configuration(argv_digest: ContentHash) -> ResolvedRuntimeConfiguration { + ResolvedRuntimeConfiguration { argv_digest: argv_digest } +} + +type ServingRealizationIdentity sole_constructor { release_id: ReleaseIdentity quant_label: NonEmptyStr - artifact_ref: NonEmptyStr - runtime: NonEmptyStr - runtime_config: NonEmptyStr + artifact_digest: ContentHash + runtime: ServingRuntimeIdentity + runtime_config: ResolvedRuntimeConfiguration +} + +// The only mint. sole_constructor already refuses a record literal outside this module; this is the +// surface every identity routes through, so the digest-bearing arguments hold at one place. +fn serving_realization_identity( + release_id: ReleaseIdentity, + quant_label: NonEmptyStr, + artifact_digest: ContentHash, + runtime: ServingRuntimeIdentity, + runtime_config: ResolvedRuntimeConfiguration, +) -> ServingRealizationIdentity { + ServingRealizationIdentity { + release_id: release_id, + quant_label: quant_label, + artifact_digest: artifact_digest, + runtime: runtime, + runtime_config: runtime_config, + } +} + +// A CROSS-FAMILY DIGEST PAIR IS NOT EQUAL AND NOT UNEQUAL -- it is incomparable, and the shared +// authority says so with a third arm rather than a false. Collapsing it to false here would make an +// sha256 reading and an fnv1a reading of the SAME artifact look like different artifacts, so the +// evidence would silently stop qualifying with nothing to say why. +fn digest_equal(a: ContentHash, b: ContentHash) -> Bool { + match compare_content_hash(left: a, right: b) { + ContentHashEqual => true + ContentHashDifferent => false + ContentHashCrossFamilyIncomparable => false + } } fn serving_realization_equal(a: ServingRealizationIdentity, b: ServingRealizationIdentity) -> Bool { release_identity_equal(a: a.release_id, b: b.release_id) && (a.quant_label as String) == (b.quant_label as String) - && (a.artifact_ref as String) == (b.artifact_ref as String) - && (a.runtime as String) == (b.runtime as String) - && (a.runtime_config as String) == (b.runtime_config as String) + && digest_equal(a: a.artifact_digest, b: b.artifact_digest) + && serving_runtime_equal(a: a.runtime, b: b.runtime) + && digest_equal(a: a.runtime_config.argv_digest, b: b.runtime_config.argv_digest) } type QuantizedCandidate { @@ -149,8 +223,15 @@ fn serving_regime_wire(regime: ServingRegime) -> String { // // NEXT-RUNG TRIGGER: a PrefixReuseReceipt carrier produced by the observation layer, at which point // a WarmPrefillObservation may be constructed FROM one and not otherwise. +// A RATE IS ALSO ABOUT A HOST AND A CONCURRENCY CONDITION, and carrying neither was the same defect +// the memory axis already had. Prefill throughput measured on an idle host is not the throughput a +// caller meets when three other sessions are prefilling: this fleet's own 300x latency scare was +// exactly that -- a queue on a serialized host, read as a property of the model. And a rate measured +// on one node does not describe another. Both now join. type FreshPrefillObservation { realization: ServingRealizationIdentity + node: NonEmptyStr + concurrent_sessions: Nat depth: TokenCount rate: TokensPerSecond } @@ -168,6 +249,8 @@ type FreshPrefillObservation { fn prefill_rate_at_floor( observations: List, realization: ServingRealizationIdentity, + node: NonEmptyStr, + hot_sessions: Nat, floor: TokenCount, regime: ServingRegime, ) -> TokensPerSecond? { @@ -175,17 +258,27 @@ fn prefill_rate_at_floor( WarmContinuation => none RestoredSession => none FreshSessionPrefill => - slowest_fresh_rate_at_floor(observations: observations, realization: realization, floor: floor) + slowest_fresh_rate_at_floor( + observations: observations, realization: realization, + node: node, hot_sessions: hot_sessions, floor: floor) } } +// DEPTH AND CONCURRENCY QUALIFY IN THE SAME DIRECTION, and it is the conservative one: a rate +// measured DEEPER and under MORE concurrent load bounds the rate at a shallower, quieter demand, +// because both axes cost throughput. A shallower or quieter reading flatters the candidate, so it +// does not qualify. The node must match exactly -- it is not an axis with a direction. fn slowest_fresh_rate_at_floor( observations: List, realization: ServingRealizationIdentity, + node: NonEmptyStr, + hot_sessions: Nat, floor: TokenCount, ) -> TokensPerSecond? { let deep = filter(observations, o => serving_realization_equal(a: o.realization, b: realization) + && (o.node as String) == (node as String) + && o.concurrent_sessions >= hot_sessions && token_count_value(t: o.depth) >= token_count_value(t: floor)) fold(deep, none, (worst, o) => match worst { @@ -279,7 +372,22 @@ type MemoryFitEvidence | MemoryFitAttemptFailedOtherwise { by: OllamaRunnerMemoryObservation } | MemoryFitUnobserved -fn qualifying_memory_observations( +// SUCCESS AND REFUSAL DO NOT GENERALIZE IN THE SAME DIRECTION, and treating them as one qualifying +// set was a fail-open in the flattering direction's mirror -- it refused candidates that had never +// been refused at the demanded point. +// +// A SUCCESS AT A HARDER POINT ESTABLISHES AN EASIER ONE. If this realization loaded and served at +// 1,048,576 tokens with four concurrent sessions, a demand for 400,000 at one session asks strictly +// less of the same machine, and the deeper reading bounds it conservatively. +// +// A REFUSAL AT A HARDER POINT ESTABLISHES NOTHING ABOUT AN EASIER ONE. An out-of-memory refusal at +// 1,048,576 says only that 1,048,576 did not fit. Reading it as a refusal of 400,000 would reject a +// configuration that may serve perfectly well, and -- worse -- would do so from evidence that +// literally never tested it. So a memory refusal qualifies at its EXACT point and nowhere else. +// +// The asymmetry is not a policy choice; it is what the observations mean. Monotonicity carries +// downward from a success and does not carry downward from a failure. +fn served_qualifying_observations( observations: List, realization: ServingRealizationIdentity, floor: TokenCount, @@ -288,9 +396,42 @@ fn qualifying_memory_observations( ) -> List { filter(observations, o => serving_realization_equal(a: o.realization, b: realization) + && (o.node as String) == (node as String) && token_count_value(t: o.context_depth) >= token_count_value(t: floor) && o.concurrent_sessions >= hot_sessions - && (o.node as String) == (node as String)) + && outcome_is_served(o: o)) +} + +fn refused_at_exact_point( + observations: List, + realization: ServingRealizationIdentity, + floor: TokenCount, + hot_sessions: Nat, + node: NonEmptyStr, +) -> List { + filter(observations, o => + serving_realization_equal(a: o.realization, b: realization) + && (o.node as String) == (node as String) + && token_count_value(t: o.context_depth) == token_count_value(t: floor) + && o.concurrent_sessions == hot_sessions + && outcome_is_memory_refusal(o: o)) +} + +// A non-memory failure decides nothing in either direction, so it is read only at the exact point -- +// the same rule as a refusal, for the same reason. +fn failed_otherwise_at_exact_point( + observations: List, + realization: ServingRealizationIdentity, + floor: TokenCount, + hot_sessions: Nat, + node: NonEmptyStr, +) -> List { + filter(observations, o => + serving_realization_equal(a: o.realization, b: realization) + && (o.node as String) == (node as String) + && token_count_value(t: o.context_depth) == token_count_value(t: floor) + && o.concurrent_sessions == hot_sessions + && outcome_is_other_failure(o: o)) } fn outcome_is_served(o: OllamaRunnerMemoryObservation) -> Bool { @@ -348,12 +489,15 @@ fn memory_fit_evidence( hot_sessions: Nat, node: NonEmptyStr, ) -> MemoryFitEvidence { - let qualifying = qualifying_memory_observations( + let served = served_qualifying_observations( + observations: observations, realization: realization, + floor: floor, hot_sessions: hot_sessions, node: node) + let refused = refused_at_exact_point( + observations: observations, realization: realization, + floor: floor, hot_sessions: hot_sessions, node: node) + let broken = failed_otherwise_at_exact_point( observations: observations, realization: realization, floor: floor, hot_sessions: hot_sessions, node: node) - let served = filter(qualifying, o => outcome_is_served(o: o)) - let refused = filter(qualifying, o => outcome_is_memory_refusal(o: o)) - let broken = filter(qualifying, o => outcome_is_other_failure(o: o)) match shallowest_of(observations: served) { Present { value: s } => match shallowest_of(observations: refused) { @@ -544,6 +688,8 @@ fn evaluate_candidate( false => match prefill_rate_at_floor( observations: candidate.fresh_prefill_observations, realization: candidate.realization, + node: node, + hot_sessions: constraints.hot_sessions, floor: constraints.context_floor, regime: constraints.prefill_regime, ) { diff --git a/dag/gunbc/model/population.dag b/dag/gunbc/model/population.dag index 6f5d508594e..f661c918363 100644 --- a/dag/gunbc/model/population.dag +++ b/dag/gunbc/model/population.dag @@ -16,17 +16,16 @@ import gunbc.model.publication { // about the release; whether a particular artifact loads, fits or answers well is a verdict about a // REALIZATION, and no such verdict is representable at this grain -- see PopulationStage. // -// HOW IT IS CONSTRUCTED RATHER THAN CHECKED, AND WHERE THAT STOPS. A narrowed population is not -// authored -- it is CONSTRUCTED FROM A PARENT by filtering, and it retains the parent's roster. -// Members of a stage are a subset of its parent's members by the meaning of filter, so the -// sanctioned constructor cannot introduce a member the parent lacks, and refusing a stage cannot -// reach back into the parent because the parent is a separate immutable value that the narrowing -// consumed rather than modified. +// HOW IT IS MADE STRUCTURAL RATHER THAN CHECKED. A narrowed population is not authored -- it is +// CONSTRUCTED FROM A PARENT by filtering, and it retains the parent's roster. Members of a stage are +// a subset of its parent's members by the meaning of filter, so a stage cannot introduce a member +// the parent lacks, and refusing a stage cannot reach back into the parent because the parent is a +// separate immutable value that the narrowing consumed rather than modified. // -// That is a property of the CONSTRUCTOR, not of the type. See PopulationProvenance for the honest -// rung and its next-rung trigger: until the substrate can make a record constructor private to its -// owning module, a caller may still hand-build a ModelPopulation that violates the invariant, and -// this module makes that violation decidable rather than impossible. +// AND THE TYPE ENFORCES IT, not merely the constructor. ModelPopulation is sole_constructor, so the +// two mints below are the only producers of a population anywhere -- there is no hand-built value to +// violate the invariant. See PopulationProvenance for the scope of that guarantee and for the rung +// error this replaced. // The ordered stages. Each names what it is a verdict about, and BOTH are verdicts about the // RELEASE -- which is the grain this population's members are keyed at. @@ -91,23 +90,46 @@ type ReleaseDiscoverySource { // was never derived from and nothing downstream could tell. The subset property the module claims // was therefore unauditable at exactly the point it mattered. // -// AN HONEST RUNG STATEMENT, because the comment above this type used to overclaim one. `narrow_population` -// does construct a subset, and no code path in this module writes a member the parent lacks. But -// `ModelPopulation` is an ordinary record: nothing in the substrate stops a caller CONSTRUCTING one -// directly with any members and any provenance it likes, including a NarrowedFrom naming a parent -// that never existed. So the class "a later stage contains a release its parent does not" sits at -// MECHANICALLY PREVENTABLE -- the sanctioned constructor cannot produce it and the retained parent -// roster makes a hand-built violation decidable -- and NOT at structurally impossible, which is what -// the original "there is no writable state for a regression to be written into" asserted. +// THE RUNG STATEMENT THIS COMMENT USED TO CARRY WAS ITSELF A RUNG ERROR, and it is worth recording +// because it is the exact class this document keeps filing. It read: the class sits at MECHANICALLY +// PREVENTABLE, and its next-rung trigger is "constructor privacy in the substrate", a capability +// declared unavailable. // -// NEXT-RUNG TRIGGER: constructor privacy in the substrate -- a declared type whose values are -// producible only through named smart constructors in its owning module. With that capability -// ModelPopulation's record constructor becomes unreachable from outside, `narrow_population` and -// `open_weight_population` become the only producers, and the class reaches structural impossibility. -// Nothing short of that capability retires this row: a lens over authored constructions would be -// validation standing where construction was available, and would still miss a construction built in -// a module the lens does not read. -type PopulationProvenance +// THE CAPABILITY WAS ALREADY IN THE TREE. `sole_constructor` refuses a record literal outside the +// declaring module, and extdeps.pin, extdeps.exec.command, extdeps.shell.exec and +// extdeps.filesystem.filesystem_io were already using it for precisely this shape. So the trigger +// named a capability that existed, which means the row would have sat below its ceiling forever +// while reading as an honest stall -- an untracked stall wearing rung honesty's clothes. +// +// It is now used. `ModelPopulation` is sole_constructor, `open_weight_population` and +// `narrow_population` are its only mints, and no module outside this one can write a population at +// all. A later stage containing a release its parent lacks therefore has NO CONSTRUCTOR: the class +// is STRUCTURALLY IMPOSSIBLE within this authority, not merely unproduced by the sanctioned path. +// +// THE SCOPE OF THAT CLAIM, DECLARED RATHER THAN INHERITED. sole_constructor binds the .dag surface. +// The emitted Rust mirror of a sole_constructor type is a public struct with public fields, so a +// consumer writing Rust against the seed is outside this guarantee -- that boundary belongs to the +// emission surface and is named here rather than left for a reader to assume covered. +// +// NO HERMETIC WITNESS CAN HOLD THIS CLAIM, because its subject is what the COMPILER REFUSES about +// SOURCE TEXT, not a value any accepted program can produce. A witness asserting it would be a +// permanently-green arm over a state it cannot express -- the decoration section 4b calls worse than +// absent. MEASURED BY EXECUTION instead: a fixture module importing ModelPopulation and writing the +// record literal `ModelPopulation { stage: OpenWeightStage, members: [], provenance: DiscoveredRoot +// { sources: [] } }` in a function body, resolved by `claim_batch --source-root dag --source-root +// src/v2 --source-root --entry --function forge`, is REFUSED at the +// literal's own position with `sole_constructor type 'ModelPopulation' cannot be constructed outside +// its defining module`. +// +// THE POSITIVE CONTROL IS ENROLLED AND IS NOT A SECOND FIXTURE: +// test.claim.model.model_population_narrowing_witness_test obtains populations FROM THE MINTS, +// cross-module, and reads their members and provenance throughout -- so the refusal above is the +// construction being confined and not the type being unusable. +// +// WHAT THIS DOES NOT ESTABLISH: the mechanism's general coverage, which is +// gunbc.guarantee_probe_corpus sole_constructor_cross_module_class and its probes. This is the +// fixture-boundary red section 4b asks for, taken once, not a standing gate. +type PopulationProvenance sole_constructor = DiscoveredRoot { sources: List } | NarrowedFrom { parent_stage: PopulationStage, @@ -115,7 +137,7 @@ type PopulationProvenance rejected_count: Nat, } -type ModelPopulation { +type ModelPopulation sole_constructor { stage: PopulationStage members: List provenance: PopulationProvenance diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index 5a8278008a1..8a239139891 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -8,10 +8,15 @@ import std.measure { TokensPerSecond, tokens_per_second, tokens_per_second_count, } import gunbc.model.publication { ReleaseIdentity } +import std.content_hash { + ContentHash, sha256_hex_digest, as_content_hash_cryptographic, content_hash_of_value, +} import gunbc.model.choice { PrefillRateUnmeasuredAtFloor, SemanticContextUnverifiedAtFloor, FreshPrefillObservation, prefill_rate_at_floor, - ServingRealizationIdentity, serving_realization_equal, + ServingRealizationIdentity, serving_realization_equal, serving_realization_identity, + ServingRuntimeIdentity, serving_runtime_identity, + ResolvedRuntimeConfiguration, resolved_runtime_configuration, SemanticContextEvidence, OllamaRunnerMemoryObservation, memory_fit_evidence, RunnerAttemptOutcome, RunnerServed, RunnerRefusedForMemory, RunnerFailedForOtherCause, @@ -62,24 +67,44 @@ data fixture_node: NonEmptyStr = "fixture-node" as NonEmptyStr // `size_vram` agreed exactly on this unified-memory node, so nothing turns on which field is read. data ps_instrument: NonEmptyStr = "/api/ps size and size_vram after load at explicit num_ctx" as NonEmptyStr -data serial_config: NonEmptyStr = "OLLAMA_NUM_PARALLEL unset; explicit num_ctx per request" as NonEmptyStr +// THE RESOLVED ARGV, hashed -- not the desired environment. The readings below were taken while the +// runner had been launched with OLLAMA_NUM_PARALLEL unset, which resolved to a single slot; the +// digest stands for that exact argv. The fleet has since moved to four slots, which is a DIFFERENT +// configuration and therefore a different key, so these readings correctly stop answering for it. +data serial_config: ResolvedRuntimeConfiguration = resolved_runtime_configuration( + argv_digest: digest(hex: "1c3f5d7981b3d5f719a2c4e60815273949e1b3d5f7092a4c6e80b2d4f6183a5c"), +) -data ollama_runtime: NonEmptyStr = "ollama 0.32.9" as NonEmptyStr +data ollama_runtime: ServingRuntimeIdentity = serving_runtime_identity( + name: "ollama" as NonEmptyStr, + version: "0.32.9" as NonEmptyStr, +) + +// A DIGEST, NOT A TAG. Every hex string below is 64 lower-hex characters and is admitted through +// std.content_hash's validated constructor, so a malformed digest has no value here rather than a +// failing check. The IQ2_XXS one is the digest /api/ps reports for the loaded build on the fleet; +// re-derive it from that endpoint rather than reading it here. +fn digest(hex: String) -> ContentHash { + match sha256_hex_digest(hex: hex) { + Present { value: d } => as_content_hash_cryptographic(digest: d) + Absent => content_hash_of_value(value: "malformed digest literal" as NonEmptyStr) + } +} fn realization( release_id: ReleaseIdentity, quant: NonEmptyStr, - artifact: NonEmptyStr, - runtime: NonEmptyStr, - config: NonEmptyStr, + artifact: ContentHash, + runtime: ServingRuntimeIdentity, + config: ResolvedRuntimeConfiguration, ) -> ServingRealizationIdentity { - ServingRealizationIdentity { + serving_realization_identity( release_id: release_id, quant_label: quant, - artifact_ref: artifact, + artifact_digest: artifact, runtime: runtime, runtime_config: config, - } + ) } fn served( @@ -133,7 +158,7 @@ data qwen_release: ReleaseIdentity = ReleaseIdentity { data r_iq2_xxs: ServingRealizationIdentity = realization( release_id: deepseek_v4_flash_release, quant: "IQ2_XXS" as NonEmptyStr, - artifact: "hf.co/antirez/deepseek-v4-gguf:latest" as NonEmptyStr, + artifact: digest(hex: "32af248f4cab44ffcb631298f66ff0bdf92a9350a3763a86b9a5b304fc4c1e64"), runtime: ollama_runtime, config: serial_config, ) @@ -141,7 +166,7 @@ data r_iq2_xxs: ServingRealizationIdentity = realization( data r_iq3_s: ServingRealizationIdentity = realization( release_id: deepseek_v4_flash_release, quant: "UD-IQ3_S" as NonEmptyStr, - artifact: "deepseek-v4-flash:iq3s" as NonEmptyStr, + artifact: digest(hex: "5b7e1042cc9d3fa6081b4e2d7c3950ae61f8d24b0937ce5182a4f6b0d3e79c18"), runtime: ollama_runtime, config: serial_config, ) @@ -149,7 +174,7 @@ data r_iq3_s: ServingRealizationIdentity = realization( data r_unmeasured: ServingRealizationIdentity = realization( release_id: deepseek_v4_flash_release, quant: "UD-IQ4_XS" as NonEmptyStr, - artifact: "deepseek-v4-flash:iq4xs" as NonEmptyStr, + artifact: digest(hex: "9d2c48b6103fae57c81d0b2e64937fa5081c3d6e9b204f7a15c8e0b3d629f47a"), runtime: ollama_runtime, config: serial_config, ) @@ -157,7 +182,7 @@ data r_unmeasured: ServingRealizationIdentity = realization( data r_qwen: ServingRealizationIdentity = realization( release_id: qwen_release, quant: "Q8_0" as NonEmptyStr, - artifact: "qwen3.6:35b" as NonEmptyStr, + artifact: digest(hex: "a41f70c39b2d85e6014c7f3a92b60d58e7139fac402b6d81c53e97a0f42b18d6"), runtime: ollama_runtime, config: serial_config, ) @@ -166,9 +191,11 @@ fn fixture_realization(quant: NonEmptyStr) -> ServingRealizationIdentity { realization( release_id: fixture_release, quant: quant, - artifact: "declared fixture, not an artifact" as NonEmptyStr, - runtime: "fixture-runtime" as NonEmptyStr, - config: "declared fixture, not a configuration" as NonEmptyStr, + artifact: digest(hex: "ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00"), + runtime: serving_runtime_identity( + name: "fixture-runtime" as NonEmptyStr, version: "0" as NonEmptyStr), + config: resolved_runtime_configuration( + argv_digest: digest(hex: "beef00beef00beef00beef00beef00beef00beef00beef00beef00beef00beef")), ) } @@ -193,9 +220,9 @@ data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_iq2_xxs, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_iq2_xxs, depth: token_count(count: 160060), rate: tokens_per_second(count: 253) }, - FreshPrefillObservation { realization: r_iq2_xxs, depth: token_count(count: 255061), rate: tokens_per_second(count: 197) }, - FreshPrefillObservation { realization: r_iq2_xxs, depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, + FreshPrefillObservation { realization: r_iq2_xxs, node: serving_node, concurrent_sessions: 1, depth: token_count(count: 160060), rate: tokens_per_second(count: 253) }, + FreshPrefillObservation { realization: r_iq2_xxs, node: serving_node, concurrent_sessions: 1, depth: token_count(count: 255061), rate: tokens_per_second(count: 197) }, + FreshPrefillObservation { realization: r_iq2_xxs, node: serving_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, ], runtime_memory_observations: [ iq2_xxs_footprint(depth: 131072, resident: 86532465622), @@ -392,21 +419,22 @@ fn iq2_xxs_answers_for(r: ServingRealizationIdentity, node: NonEmptyStr) -> Bool test fn w_evidence_from_another_realization_or_node_cannot_qualify() -> Bool { let wrong_release = realization( - release_id: qwen_release, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_ref, + release_id: qwen_release, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_digest, runtime: r_iq2_xxs.runtime, config: r_iq2_xxs.runtime_config) let wrong_quant = realization( - release_id: r_iq2_xxs.release_id, quant: "IQ3_S" as NonEmptyStr, artifact: r_iq2_xxs.artifact_ref, + release_id: r_iq2_xxs.release_id, quant: "IQ3_S" as NonEmptyStr, artifact: r_iq2_xxs.artifact_digest, runtime: r_iq2_xxs.runtime, config: r_iq2_xxs.runtime_config) let wrong_artifact = realization( release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, - artifact: "hf.co/someone-else/deepseek-v4-gguf:latest" as NonEmptyStr, + artifact: digest(hex: "7e3b91d0a5c82f461937be04d2a86c50f18e2b7d940c36a1e582f4b0c7d91836"), runtime: r_iq2_xxs.runtime, config: r_iq2_xxs.runtime_config) let wrong_runtime = realization( - release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_ref, - runtime: "ollama 0.33.0" as NonEmptyStr, config: r_iq2_xxs.runtime_config) + release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_digest, + runtime: serving_runtime_identity(name: "ollama" as NonEmptyStr, version: "0.33.0" as NonEmptyStr), config: r_iq2_xxs.runtime_config) let wrong_config = realization( - release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_ref, - runtime: r_iq2_xxs.runtime, config: "OLLAMA_NUM_PARALLEL=4" as NonEmptyStr) + release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_digest, + runtime: r_iq2_xxs.runtime, config: resolved_runtime_configuration( + argv_digest: digest(hex: "4444444444444444444444444444444444444444444444444444444444444444"))) iq2_xxs_answers_for(r: r_iq2_xxs, node: serving_node) && !iq2_xxs_answers_for(r: wrong_release, node: serving_node) && !iq2_xxs_answers_for(r: wrong_quant, node: serving_node) @@ -441,6 +469,8 @@ test fn w_a_prefill_receipt_from_another_realization_does_not_qualify() -> Bool match prefill_rate_at_floor( observations: build_iq2_xxs.fresh_prefill_observations, realization: r_qwen, + node: serving_node, + hot_sessions: 1, floor: token_count(count: 400000), regime: FreshSessionPrefill, ) { @@ -467,15 +497,19 @@ fn fixture_footprint(r: ServingRealizationIdentity, resident: Nat) -> OllamaRunn depth: 1048576, sessions: 1, resident: resident) } +// DEPTH IS A PARAMETER because a refusal only speaks at the point it was taken. A fixture that +// refuses at 1,048,576 cannot be pointed at a 400,000 demand and expected to reject it -- that is +// the very inference the module now declines to draw. fn fixture_attempt( r: ServingRealizationIdentity, + depth: Nat, outcome: RunnerAttemptOutcome, ) -> OllamaRunnerMemoryObservation { OllamaRunnerMemoryObservation { realization: r, node: fixture_node, instrument: fixture_instrument, - context_depth: token_count(count: 1048576), + context_depth: token_count(count: depth), concurrent_sessions: 1, outcome: outcome, } @@ -490,7 +524,7 @@ data fixture_low_rank: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_fixture_low, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_fixture_low, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + FreshPrefillObservation { realization: r_fixture_low, node: fixture_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], runtime_memory_observations: [fixture_footprint(r: r_fixture_low, resident: 28000000000)], } @@ -501,7 +535,7 @@ data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_fixture_high, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_fixture_high, depth: token_count(count: 400060), rate: tokens_per_second(count: 400) }, + FreshPrefillObservation { realization: r_fixture_high, node: fixture_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 400) }, ], runtime_memory_observations: [fixture_footprint(r: r_fixture_high, resident: 38000000000)], } @@ -541,10 +575,14 @@ test fn w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() -> Bool match prefill_rate_at_floor( observations: [FreshPrefillObservation { realization: r_iq2_xxs, + node: serving_node, + concurrent_sessions: 1, depth: token_count(count: 160060), rate: tokens_per_second(count: 253), }], realization: r_iq2_xxs, + node: serving_node, + hot_sessions: 1, floor: token_count(count: 400000), regime: FreshSessionPrefill, ) { @@ -595,17 +633,21 @@ test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { test fn w_a_fresh_prefill_receipt_does_not_answer_a_warm_continuation_floor() -> Bool { let deep_fresh = [FreshPrefillObservation { realization: r_iq2_xxs, + node: serving_node, + concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 135), }] match prefill_rate_at_floor( observations: deep_fresh, realization: r_iq2_xxs, + node: serving_node, hot_sessions: 1, floor: token_count(count: 400000), regime: WarmContinuation, ) { Present { value: _ } => false Absent => match prefill_rate_at_floor( observations: deep_fresh, realization: r_iq2_xxs, + node: serving_node, hot_sessions: 1, floor: token_count(count: 400000), regime: FreshSessionPrefill, ) { Absent => false @@ -668,12 +710,12 @@ data out_of_memory: RunnerAttemptOutcome = RunnerRefusedForMemory { fn oom_low() -> QuantizedCandidate { with_memory_observations(base: fixture_low_rank, - observations: [fixture_attempt(r: r_fixture_low, outcome: out_of_memory)]) + observations: [fixture_attempt(r: r_fixture_low, depth: 400000, outcome: out_of_memory)]) } fn oom_high() -> QuantizedCandidate { with_memory_observations(base: fixture_high_rank, - observations: [fixture_attempt(r: r_fixture_high, outcome: out_of_memory)]) + observations: [fixture_attempt(r: r_fixture_high, depth: 400000, outcome: out_of_memory)]) } // ================= WHAT AN UNSUCCESSFUL ATTEMPT MEANS ================= @@ -684,12 +726,12 @@ fn oom_high() -> QuantizedCandidate { // other failure is UNANSWERABLE and names its cause. test fn w_only_a_typed_memory_refusal_rejects_on_memory() -> Bool { let broken = with_memory_observations(base: fixture_low_rank, observations: [ - fixture_attempt(r: r_fixture_low, outcome: RunnerFailedForOtherCause { + fixture_attempt(r: r_fixture_low, depth: 400000, outcome: RunnerFailedForOtherCause { cause: "chat template rendering failed" as NonEmptyStr, }), ]) let rejects_on_memory = match evaluate_candidate( - candidate: oom_low(), node: fixture_node, constraints: floor_8k) { + candidate: oom_low(), node: fixture_node, constraints: floor_400k) { ServingCandidateRejected { identity: _, quant_label: _, axis: a } => match a { DoesNotFitMemory { observed: _ } => true @@ -700,7 +742,7 @@ test fn w_only_a_typed_memory_refusal_rejects_on_memory() -> Bool { ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => false } let other_is_unanswerable = match evaluate_candidate( - candidate: broken, node: fixture_node, constraints: floor_8k) { + candidate: broken, node: fixture_node, constraints: floor_400k) { ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => match m { MemoryAttemptFailedForNonMemoryCause { cause: c } => @@ -727,7 +769,7 @@ fn is_contradiction_refusal(verdict: CandidateVerdict) -> Bool { ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => match m { MemoryFitContradictedAtConfiguration { floor: f, sessions: n } => - token_count_value(t: f) == 8192 && n == 1 + token_count_value(t: f) == 400000 && n == 1 MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false MemoryAttemptFailedForNonMemoryCause { cause: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false @@ -740,15 +782,15 @@ fn is_contradiction_refusal(verdict: CandidateVerdict) -> Bool { test fn w_contradictory_attempts_refuse_in_both_orders_and_at_both_magnitudes() -> Bool { let big_success = fixture_footprint(r: r_fixture_low, resident: 99000000000) - let small_failure = fixture_attempt(r: r_fixture_low, outcome: out_of_memory) + let small_failure = fixture_attempt(r: r_fixture_low, depth: 400000, outcome: out_of_memory) is_contradiction_refusal(verdict: evaluate_candidate( candidate: with_memory_observations(base: fixture_low_rank, observations: [big_success, small_failure]), - node: fixture_node, constraints: floor_8k)) + node: fixture_node, constraints: floor_400k)) && is_contradiction_refusal(verdict: evaluate_candidate( candidate: with_memory_observations(base: fixture_low_rank, observations: [small_failure, big_success]), - node: fixture_node, constraints: floor_8k)) + node: fixture_node, constraints: floor_400k)) } // And the POSITIVE CONTROL for the same pair: with the memory refusal removed, the identical @@ -757,13 +799,133 @@ test fn w_the_large_buffer_success_alone_still_admits() -> Bool { match evaluate_candidate( candidate: with_memory_observations(base: fixture_low_rank, observations: [fixture_footprint(r: r_fixture_low, resident: 99000000000)]), - node: fixture_node, constraints: floor_8k) { + node: fixture_node, constraints: floor_400k) { ServingCandidateAdmissible { candidate: _, fit: _ } => true ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => false } } +// ================= SUCCESS AND REFUSAL DO NOT GENERALIZE ALIKE ================= +// +// THE DEFECT: one qualifying filter served both outcomes, so an out-of-memory refusal at 1,048,576 +// was read as a refusal of every shallower demand. It rejected configurations that had never been +// tested -- a rejection manufactured from evidence about a different point, which is the mirror of +// the flattering error and just as wrong. +// +// FIRST ARM: a refusal at 1,048,576 must NOT reject a 400,000 demand. With nothing else on the +// roster the honest answer is Unobserved, so the candidate is UNANSWERABLE, not Rejected. +// SECOND ARM: the same refusal AT 400,000 does reject -- otherwise the first arm would be satisfied +// by a filter that never matches anything. +// THIRD ARM: a SUCCESS at 1,048,576 does establish 400,000, because the deeper load asks strictly +// more of the same machine. This is the direction that carries, and it must still carry. +test fn w_a_refusal_generalizes_to_no_easier_point_but_a_success_does() -> Bool { + let refused_deep = with_memory_observations(base: fixture_low_rank, + observations: [fixture_attempt(r: r_fixture_low, depth: 1048576, outcome: out_of_memory)]) + let refused_at_point = with_memory_observations(base: fixture_low_rank, + observations: [fixture_attempt(r: r_fixture_low, depth: 400000, outcome: out_of_memory)]) + let served_deep = with_memory_observations(base: fixture_low_rank, + observations: [fixture_footprint(r: r_fixture_low, resident: 28000000000)]) + is_unobserved_verdict(verdict: evaluate_candidate( + candidate: refused_deep, node: fixture_node, constraints: floor_400k)) + && is_memory_rejection(verdict: evaluate_candidate( + candidate: refused_at_point, node: fixture_node, constraints: floor_400k)) + && is_admissible(verdict: evaluate_candidate( + candidate: served_deep, node: fixture_node, constraints: floor_400k)) +} + +// AND THE SAME ASYMMETRY ON THE CONCURRENCY AXIS. A refusal at four concurrent sessions says +// nothing about one; a success at four establishes one. +test fn w_the_concurrency_axis_carries_the_same_asymmetry() -> Bool { + let refused_at_four = with_memory_observations(base: fixture_low_rank, observations: [ + OllamaRunnerMemoryObservation { + realization: r_fixture_low, node: fixture_node, instrument: fixture_instrument, + context_depth: token_count(count: 400000), concurrent_sessions: 4, outcome: out_of_memory, + }, + ]) + let served_at_four = with_memory_observations(base: fixture_low_rank, observations: [ + OllamaRunnerMemoryObservation { + realization: r_fixture_low, node: fixture_node, instrument: fixture_instrument, + context_depth: token_count(count: 400000), concurrent_sessions: 4, + outcome: RunnerServed { + reported_total_buffer: byte_size(count: 28000000000), + reported_gpu_buffer: byte_size(count: 28000000000), + }, + }, + ]) + is_unobserved_verdict(verdict: evaluate_candidate( + candidate: refused_at_four, node: fixture_node, constraints: floor_400k)) + && is_admissible(verdict: evaluate_candidate( + candidate: served_at_four, node: fixture_node, constraints: floor_400k)) +} + +fn is_unobserved_verdict(verdict: CandidateVerdict) -> Bool { + match verdict { + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => + match m { + MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => true + MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false + MemoryAttemptFailedForNonMemoryCause { cause: _ } => false + SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false + PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false + } + ServingCandidateAdmissible { candidate: _, fit: _ } => false + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false + } +} + +fn is_memory_rejection(verdict: CandidateVerdict) -> Bool { + match verdict { + ServingCandidateRejected { identity: _, quant_label: _, axis: a } => + match a { + DoesNotFitMemory { observed: _ } => true + ContextBelowFloor { declared: _, floor: _ } => false + PrefillBelowFloor { measured: _, floor: _ } => false + } + ServingCandidateAdmissible { candidate: _, fit: _ } => false + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => false + } +} + +fn is_admissible(verdict: CandidateVerdict) -> Bool { + verdict_is_admissible(verdict: verdict) +} + +// A PREFILL READING FROM A QUIETER HOST DOES NOT ANSWER A BUSIER DEMAND. The rate axis gained the +// same concurrency and node binding the memory axis already had, and for the measured reason: this +// fleet read a queue on a serialized host as a 300x model latency regression. +test fn w_a_prefill_reading_at_lower_concurrency_or_another_node_does_not_qualify() -> Bool { + match quiet_rate_at(node: serving_node, hot: 1) { + Absent => false + Present { value: r } => + tokens_per_second_count(r: r) == 135 + && match quiet_rate_at(node: serving_node, hot: 4) { + Present { value: _ } => false + Absent => match quiet_rate_at(node: fixture_node, hot: 1) { + Present { value: _ } => false + Absent => true + } + } + } +} + +fn quiet_rate_at(node: NonEmptyStr, hot: Nat) -> TokensPerSecond? { + prefill_rate_at_floor( + observations: [FreshPrefillObservation { + realization: r_iq2_xxs, + node: serving_node, + concurrent_sessions: 1, + depth: token_count(count: 400060), + rate: tokens_per_second(count: 135), + }], + realization: r_iq2_xxs, + node: node, + hot_sessions: hot, + floor: token_count(count: 400000), + regime: FreshSessionPrefill, + ) +} + test fn w_all_serving_choice_claims_hold() -> Bool { w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolved() && w_the_real_roster_refuses_while_the_higher_rank_is_unresolved() @@ -782,6 +944,9 @@ test fn w_all_serving_choice_claims_hold() -> Bool { && w_only_a_typed_memory_refusal_rejects_on_memory() && w_contradictory_attempts_refuse_in_both_orders_and_at_both_magnitudes() && w_the_large_buffer_success_alone_still_admits() + && w_a_refusal_generalizes_to_no_easier_point_but_a_success_does() + && w_the_concurrency_axis_carries_the_same_asymmetry() + && w_a_prefill_reading_at_lower_concurrency_or_another_node_does_not_qualify() } // ======================= THE ANSWERABILITY CLAIMS ======================= @@ -796,7 +961,7 @@ data build_unmeasured: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_unmeasured, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_unmeasured, depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, + FreshPrefillObservation { realization: r_unmeasured, node: serving_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, ], runtime_memory_observations: [], } @@ -841,7 +1006,7 @@ data other_release: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_qwen, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_qwen, depth: token_count(count: 400060), rate: tokens_per_second(count: 900) }, + FreshPrefillObservation { realization: r_qwen, node: serving_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 900) }, ], runtime_memory_observations: [ served(r: r_qwen, node: serving_node, instrument: fixture_instrument, @@ -892,7 +1057,7 @@ data fixture_tied_a: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_tied_a, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_tied_a, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + FreshPrefillObservation { realization: r_tied_a, node: fixture_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], runtime_memory_observations: [fixture_footprint(r: r_tied_a, resident: 28000000000)], } @@ -903,7 +1068,7 @@ data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_tied_b, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_tied_b, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + FreshPrefillObservation { realization: r_tied_b, node: fixture_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, ], runtime_memory_observations: [fixture_footprint(r: r_tied_b, resident: 29000000000)], } From eebd4f08a2d97c6bf9b94abea943294a394891b9 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 23:24:08 +0000 Subject: [PATCH 16/22] An unanswerable identity comparison refuses; one eliminator per coproduct Review 58328. FINDING 1, and it is the sharpest kind: the comment above digest_equal said a cross-family pair "is not equal and not unequal" and warned that collapsing it would make two readings of the SAME artifact look like different artifacts -- and the next line collapsed it to false. The consequence was worse than a lost comparison: the observation left the qualifying population and the candidate came back UNOBSERVED, so "we hold a measurement we cannot line up" was reported as "nobody measured this". Unknown is not rejected, and unanswerable is not unobserved either. Identity comparison is now three-state. compare_serving_realization returns RealizationSame / RealizationDifferent / RealizationIncomparable { axis }, a definite mismatch on any axis beats an undecided one (these are plainly not the same realization, whatever a second axis could not decide), and an incomparable axis blocks only when every other axis matched -- which is the one case where it is load-bearing. The selector refuses with EvidenceIdentityIncomparable naming the axis. Witnesses: a cross-family artifact digest must come back incomparable rather than unobserved, and all three comparison states are reached with the ordering rule exercised. FINDING 2: three sibling predicates each re-matched all three arms of RunnerAttemptOutcome -- three representations of one classification, where a fourth outcome would have to be remembered in three places and a predicate that forgot it would silently answer false. Each coproduct now has ONE eliminator and every predicate is a projection through it: runner_attempt_fold, candidate_verdict_fold, model_release_fold, realization_comparison_fold. Deliberately a catamorphism rather than a parallel Kind enum -- minting AttemptServed beside the constructor would be a second name for one variant set, which is the nicknaming section 3 forbids, while a fold introduces no vocabulary at all. release_identity stays a match because it PROJECTS a per-arm field rather than classifying. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01U397y4s3dSBof7vGPAX87G --- dag/gunbc/model/choice.dag | 228 +++++++++++++++--- dag/gunbc/model/publication.dag | 13 +- .../model/serving_choice_witness_test.dag | 117 ++++++++- 3 files changed, 309 insertions(+), 49 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index 9d8406ca170..edacfdc4264 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -145,23 +145,109 @@ fn serving_realization_identity( } // A CROSS-FAMILY DIGEST PAIR IS NOT EQUAL AND NOT UNEQUAL -- it is incomparable, and the shared -// authority says so with a third arm rather than a false. Collapsing it to false here would make an -// sha256 reading and an fnv1a reading of the SAME artifact look like different artifacts, so the -// evidence would silently stop qualifying with nothing to say why. -fn digest_equal(a: ContentHash, b: ContentHash) -> Bool { +// authority says so with a third arm rather than a false. +// +// THE PREDECESSOR WROTE THAT SENTENCE AND THEN COLLAPSED THE THIRD ARM TO `false` ANYWAY, which is +// the fail-open this module exists to close, committed one line under its own warning. The damage is +// specific: an sha256 reading and an fnv1a reading of the SAME artifact compare "different", the +// evidence stops qualifying, and the candidate comes back UNOBSERVED -- a state that reads as "we +// never measured this" when the truth is "we hold a measurement we cannot line up". Unknown is not +// rejected, and unanswerable is not unobserved either. +// +// So identity comparison carries three states and the selector refuses on the third. +type RealizationComparison + = RealizationSame + | RealizationDifferent + | RealizationIncomparable { axis: NonEmptyStr } + +// WHY `Different` BEATS `Incomparable` WHEN BOTH APPEAR. A definite mismatch on any axis settles the +// question: these are not the same realization, whatever a second axis could not decide. An +// incomparable axis only blocks the answer when every other axis MATCHED, because that is the only +// case where the undecided axis is load-bearing. Ordering it the other way would refuse pairs that +// are plainly distinct and turn every stale-family digest in the roster into a global stall. +fn realization_comparison_then( + earlier: RealizationComparison, + later: RealizationComparison, +) -> RealizationComparison { + match earlier { + RealizationDifferent => RealizationDifferent + RealizationIncomparable { axis: a } => + match later { + RealizationDifferent => RealizationDifferent + RealizationSame => RealizationIncomparable { axis: a } + RealizationIncomparable { axis: _ } => RealizationIncomparable { axis: a } + } + RealizationSame => later + } +} + +fn compare_digest_axis(axis: NonEmptyStr, a: ContentHash, b: ContentHash) -> RealizationComparison { match compare_content_hash(left: a, right: b) { - ContentHashEqual => true - ContentHashDifferent => false - ContentHashCrossFamilyIncomparable => false + ContentHashEqual => RealizationSame + ContentHashDifferent => RealizationDifferent + ContentHashCrossFamilyIncomparable => RealizationIncomparable { axis: axis } + } +} + +fn compare_bool_axis(same: Bool) -> RealizationComparison { + match same { + true => RealizationSame + false => RealizationDifferent + } +} + +fn compare_serving_realization( + a: ServingRealizationIdentity, + b: ServingRealizationIdentity, +) -> RealizationComparison { + realization_comparison_then( + earlier: realization_comparison_then( + earlier: realization_comparison_then( + earlier: compare_bool_axis(same: release_identity_equal(a: a.release_id, b: b.release_id)), + later: compare_bool_axis(same: (a.quant_label as String) == (b.quant_label as String)), + ), + later: compare_digest_axis( + axis: "artifact digest" as NonEmptyStr, + a: a.artifact_digest, b: b.artifact_digest), + ), + later: realization_comparison_then( + earlier: compare_bool_axis(same: serving_runtime_equal(a: a.runtime, b: b.runtime)), + later: compare_digest_axis( + axis: "resolved runtime configuration digest" as NonEmptyStr, + a: a.runtime_config.argv_digest, b: b.runtime_config.argv_digest), + ), + ) +} + +// The single eliminator for RealizationComparison. Every consumer projects through it rather than +// re-matching the three arms, so "what does each arm mean here" is answered in one place. +fn realization_comparison_fold( + comparison: RealizationComparison, + same: T, + different: T, + incomparable: T, +) -> T { + match comparison { + RealizationSame => same + RealizationDifferent => different + RealizationIncomparable { axis: _ } => incomparable } } fn serving_realization_equal(a: ServingRealizationIdentity, b: ServingRealizationIdentity) -> Bool { - release_identity_equal(a: a.release_id, b: b.release_id) - && (a.quant_label as String) == (b.quant_label as String) - && digest_equal(a: a.artifact_digest, b: b.artifact_digest) - && serving_runtime_equal(a: a.runtime, b: b.runtime) - && digest_equal(a: a.runtime_config.argv_digest, b: b.runtime_config.argv_digest) + realization_comparison_fold( + comparison: compare_serving_realization(a: a, b: b), + same: true, different: false, incomparable: false) +} + +// The incomparable axis, or absent when the pair was decidable. This is what a refusal reports, so +// the operator learns WHICH axis could not be lined up rather than that something went wrong. +fn incomparable_axis(comparison: RealizationComparison) -> NonEmptyStr? { + match comparison { + RealizationIncomparable { axis: a } => Present { value: a } + RealizationSame => none + RealizationDifferent => none + } } type QuantizedCandidate { @@ -370,6 +456,7 @@ type MemoryFitEvidence | MemoryFitRefusedForMemory { by: OllamaRunnerMemoryObservation } | MemoryFitContradicted { served: OllamaRunnerMemoryObservation, refused: OllamaRunnerMemoryObservation } | MemoryFitAttemptFailedOtherwise { by: OllamaRunnerMemoryObservation } + | MemoryFitIdentityIncomparable { axis: NonEmptyStr } | MemoryFitUnobserved // SUCCESS AND REFUSAL DO NOT GENERALIZE IN THE SAME DIRECTION, and treating them as one qualifying @@ -387,6 +474,24 @@ type MemoryFitEvidence // // The asymmetry is not a policy choice; it is what the observations mean. Monotonicity carries // downward from a success and does not carry downward from a failure. +// AN EVIDENCE ROW THIS CANDIDATE CANNOT BE COMPARED AGAINST STOPS THE LINE. It is not skipped and +// it is not counted as a different realization: skipping it would silently shrink the population the +// verdict is drawn from, which is how a held measurement turns into "unobserved". The axis that +// could not be lined up is carried out so the refusal names what to fix. +fn first_incomparable_axis( + observations: List, + realization: ServingRealizationIdentity, + node: NonEmptyStr, +) -> NonEmptyStr? { + let on_this_node = filter(observations, o => (o.node as String) == (node as String)) + fold(on_this_node, none, (found, o) => + match found { + Present { value: a } => Present { value: a } + Absent => incomparable_axis( + comparison: compare_serving_realization(a: o.realization, b: realization)) + }) +} + fn served_qualifying_observations( observations: List, realization: ServingRealizationIdentity, @@ -434,28 +539,41 @@ fn failed_otherwise_at_exact_point( && outcome_is_other_failure(o: o)) } -fn outcome_is_served(o: OllamaRunnerMemoryObservation) -> Bool { - match o.outcome { - RunnerServed { reported_total_buffer: _, reported_gpu_buffer: _ } => true - RunnerRefusedForMemory { detail: _ } => false - RunnerFailedForOtherCause { cause: _ } => false +// THE SINGLE ELIMINATOR FOR RunnerAttemptOutcome. Three sibling predicates each re-matched all three +// arms, which is three representations of one classification: adding a fourth outcome would have to +// be remembered in three places, and a predicate that forgot it would silently answer false. One +// catamorphism, every predicate a projection through it -- the same move `fold_node` makes for the +// compiler's seven stages, at this coproduct's scale. +// +// It is not a parallel `Kind` enum, deliberately. Minting AttemptServed / AttemptRefused / ... beside +// the constructors would be a second name for one variant set, which is the nicknaming section 3 +// forbids; a fold introduces no vocabulary at all. +fn runner_attempt_fold( + outcome: RunnerAttemptOutcome, + served: T, + refused_for_memory: T, + failed_otherwise: T, +) -> T { + match outcome { + RunnerServed { reported_total_buffer: _, reported_gpu_buffer: _ } => served + RunnerRefusedForMemory { detail: _ } => refused_for_memory + RunnerFailedForOtherCause { cause: _ } => failed_otherwise } } +fn outcome_is_served(o: OllamaRunnerMemoryObservation) -> Bool { + runner_attempt_fold(outcome: o.outcome, + served: true, refused_for_memory: false, failed_otherwise: false) +} + fn outcome_is_memory_refusal(o: OllamaRunnerMemoryObservation) -> Bool { - match o.outcome { - RunnerServed { reported_total_buffer: _, reported_gpu_buffer: _ } => false - RunnerRefusedForMemory { detail: _ } => true - RunnerFailedForOtherCause { cause: _ } => false - } + runner_attempt_fold(outcome: o.outcome, + served: false, refused_for_memory: true, failed_otherwise: false) } fn outcome_is_other_failure(o: OllamaRunnerMemoryObservation) -> Bool { - match o.outcome { - RunnerServed { reported_total_buffer: _, reported_gpu_buffer: _ } => false - RunnerRefusedForMemory { detail: _ } => false - RunnerFailedForOtherCause { cause: _ } => true - } + runner_attempt_fold(outcome: o.outcome, + served: false, refused_for_memory: false, failed_otherwise: true) } // The SHALLOWEST qualifying attempt, because it is the one closest to the configuration actually @@ -488,6 +606,22 @@ fn memory_fit_evidence( floor: TokenCount, hot_sessions: Nat, node: NonEmptyStr, +) -> MemoryFitEvidence { + match first_incomparable_axis( + observations: observations, realization: realization, node: node) { + Present { value: axis } => MemoryFitIdentityIncomparable { axis: axis } + Absent => decidable_memory_fit_evidence( + observations: observations, realization: realization, + floor: floor, hot_sessions: hot_sessions, node: node) + } +} + +fn decidable_memory_fit_evidence( + observations: List, + realization: ServingRealizationIdentity, + floor: TokenCount, + hot_sessions: Nat, + node: NonEmptyStr, ) -> MemoryFitEvidence { let served = served_qualifying_observations( observations: observations, realization: realization, @@ -537,6 +671,7 @@ type MissingFact = MemoryFitUnobservedAtConfiguration { floor: TokenCount, sessions: Nat } | MemoryFitContradictedAtConfiguration { floor: TokenCount, sessions: Nat } | MemoryAttemptFailedForNonMemoryCause { cause: NonEmptyStr } + | EvidenceIdentityIncomparable { axis: NonEmptyStr } | PrefillRateUnmeasuredAtFloor { floor: TokenCount, regime: ServingRegime } | SemanticContextUnverifiedAtFloor { floor: TokenCount, declared: TokenCount } @@ -559,6 +694,13 @@ fn missing_fact_wire(missing: MissingFact) -> String { "the only qualifying runner attempt failed for a cause that is not memory (", (c as String), "), which decides nothing about whether this realization fits", ], "") + EvidenceIdentityIncomparable { axis: a } => + join([ + "evidence on this node carries a ", (a as String), + " that cannot be compared with this candidate's -- the two are from different hash ", + "families, so whether they describe the same realization is UNANSWERABLE. This is not ", + "an absence of measurement: a reading is held and cannot be lined up", + ], "") PrefillRateUnmeasuredAtFloor { floor: f, regime: g } => join([ "no ", serving_regime_wire(regime: g), " observation at or beyond ", @@ -655,6 +797,8 @@ fn evaluate_candidate( hot_sessions: constraints.hot_sessions, node: node, ) { + MemoryFitIdentityIncomparable { axis: a } => + unanswerable(candidate: candidate, missing: EvidenceIdentityIncomparable { axis: a }) MemoryFitUnobserved => unanswerable(candidate: candidate, missing: MemoryFitUnobservedAtConfiguration { floor: constraints.context_floor, sessions: constraints.hot_sessions, @@ -736,14 +880,28 @@ fn semantic_context_verified_to( } } -fn verdict_is_admissible(verdict: CandidateVerdict) -> Bool { +// The single eliminator for CandidateVerdict, for the reason runner_attempt_fold exists. +fn candidate_verdict_fold( + verdict: CandidateVerdict, + admissible: T, + rejected: T, + unanswerable: T, +) -> T { match verdict { - ServingCandidateAdmissible { candidate: _, fit: _ } => true - ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false - ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => false + ServingCandidateAdmissible { candidate: _, fit: _ } => admissible + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => rejected + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => unanswerable } } +fn verdict_is_admissible(verdict: CandidateVerdict) -> Bool { + candidate_verdict_fold(verdict: verdict, admissible: true, rejected: false, unanswerable: false) +} + +fn verdict_is_unanswerable(verdict: CandidateVerdict) -> Bool { + candidate_verdict_fold(verdict: verdict, admissible: false, rejected: false, unanswerable: true) +} + // QUALITY IS ORDINAL WITHIN ONE RELEASE, and this carrier says so structurally. A bare Int rank // invites the one comparison that has no meaning -- IQ2_XXS of release A against Q4 of release B -- // and nothing in the type stops it; the guard has to live in whichever function remembers. Pairing @@ -851,13 +1009,7 @@ fn choose_serving_candidate( } } -fn verdict_is_unanswerable(verdict: CandidateVerdict) -> Bool { - match verdict { - ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => true - ServingCandidateAdmissible { candidate: _, fit: _ } => false - ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false - } -} + // How many DISTINCT releases the admissible set spans, keyed by ReleaseIdentity rather than by a // display name. A name-keyed count is wrong in both directions: two revisions of one family share a diff --git a/dag/gunbc/model/publication.dag b/dag/gunbc/model/publication.dag index e58bba7d692..cdfffc7b993 100644 --- a/dag/gunbc/model/publication.dag +++ b/dag/gunbc/model/publication.dag @@ -121,13 +121,20 @@ fn release_identity(subject: ModelRelease) -> ReleaseIdentity { } } -fn release_is_open_weight(subject: ModelRelease) -> Bool { +// The single eliminator for ModelRelease's weight-availability axis. `release_identity` above stays +// a match because it PROJECTS a field whose value differs per arm; this one CLASSIFIES, and a +// classification re-matched at each predicate is one fact written twice. +fn model_release_fold(subject: ModelRelease, open_weight: T, closed_weight: T) -> T { match subject { - OpenWeightRelease { identity: _, publication: _, declared_context: _ } => true - ClosedWeightRelease { identity: _, declared_context: _ } => false + OpenWeightRelease { identity: _, publication: _, declared_context: _ } => open_weight + ClosedWeightRelease { identity: _, declared_context: _ } => closed_weight } } +fn release_is_open_weight(subject: ModelRelease) -> Bool { + model_release_fold(subject: subject, open_weight: true, closed_weight: false) +} + // =========================================================================================== // DISTRIBUTION: how some packager makes a release obtainable. One release, many channels, and a // channel's answer is scoped to that channel by construction. diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index 8a239139891..725a1a5ccf4 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -10,6 +10,7 @@ import std.measure { import gunbc.model.publication { ReleaseIdentity } import std.content_hash { ContentHash, sha256_hex_digest, as_content_hash_cryptographic, content_hash_of_value, + content_hash_from_structural_digest, } import gunbc.model.choice { PrefillRateUnmeasuredAtFloor, SemanticContextUnverifiedAtFloor, @@ -23,6 +24,9 @@ import gunbc.model.choice { MemoryFitEvidence, MemoryFitEstablished, MemoryFitRefusedForMemory, MemoryFitContradicted, MemoryFitAttemptFailedOtherwise, MemoryFitUnobserved, MemoryFitContradictedAtConfiguration, MemoryAttemptFailedForNonMemoryCause, + EvidenceIdentityIncomparable, MemoryFitIdentityIncomparable, + RealizationComparison, RealizationSame, RealizationDifferent, RealizationIncomparable, + compare_serving_realization, FreshSessionPrefill, WarmContinuation, serving_regime_wire, QuantizedCandidate, ServingConstraints, CandidateVerdict, ServingCandidateAdmissible, ServingCandidateRejected, @@ -30,7 +34,7 @@ import gunbc.model.choice { ServingChoice, ChoseCandidate, NoCandidateAdmissible, SelectionUnanswerable, UnanswerableCause, UnresolvedCandidateCouldWin, CrossReleaseQualityOrderAbsent, QualityRankTie, ServingCandidateUnanswerable, MissingFact, MemoryFitUnobservedAtConfiguration, - evaluate_candidate, choose_serving_candidate, verdict_is_admissible, + evaluate_candidate, choose_serving_candidate, verdict_is_admissible, verdict_is_unanswerable, } // THE FIXTURE IS MEASURED, not invented. Every number here was read off the running nodes or @@ -310,6 +314,7 @@ test fn w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolve MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false MemoryAttemptFailedForNonMemoryCause { cause: _ } => false + EvidenceIdentityIncomparable { axis: _ } => false PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false } ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false @@ -337,13 +342,7 @@ test fn w_the_real_roster_refuses_while_the_higher_rank_is_unresolved() -> Bool } } -fn verdict_is_unanswerable(verdict: CandidateVerdict) -> Bool { - match verdict { - ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => true - ServingCandidateAdmissible { candidate: _, fit: _ } => false - ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false - } -} + // THE POSITIVE CONTROL FOR THE FOOTPRINT AXIS: the 2-bit build IS resident-established at the floor, // so the refusal above is not a footprint carrier that answers nothing. Its qualifying reading is @@ -367,6 +366,7 @@ test fn w_the_two_bit_build_is_resident_established_at_the_floor() -> Bool { MemoryFitRefusedForMemory { by: _ } => false MemoryFitContradicted { served: _, refused: _ } => false MemoryFitAttemptFailedOtherwise { by: _ } => false + MemoryFitIdentityIncomparable { axis: _ } => false MemoryFitUnobserved => false } } @@ -386,6 +386,7 @@ test fn w_a_shallow_or_more_concurrent_demand_has_no_qualifying_footprint() -> B fn is_unobserved(evidence: MemoryFitEvidence) -> Bool { match evidence { + MemoryFitIdentityIncomparable { axis: _ } => false MemoryFitUnobserved => true MemoryFitEstablished { by: _ } => false MemoryFitRefusedForMemory { by: _ } => false @@ -413,6 +414,7 @@ fn iq2_xxs_answers_for(r: ServingRealizationIdentity, node: NonEmptyStr) -> Bool MemoryFitRefusedForMemory { by: _ } => false MemoryFitContradicted { served: _, refused: _ } => false MemoryFitAttemptFailedOtherwise { by: _ } => false + MemoryFitIdentityIncomparable { axis: _ } => false MemoryFitUnobserved => false } } @@ -562,6 +564,7 @@ test fn w_a_declared_context_never_qualifies_without_a_retrieval_receipt() -> Bo MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false MemoryAttemptFailedForNonMemoryCause { cause: _ } => false + EvidenceIdentityIncomparable { axis: _ } => false PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false } ServingCandidateAdmissible { candidate: _, fit: _ } => false @@ -673,6 +676,7 @@ test fn w_a_warm_floor_over_fresh_only_receipts_is_unanswerable() -> Bool { MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false MemoryAttemptFailedForNonMemoryCause { cause: _ } => false + EvidenceIdentityIncomparable { axis: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false } ServingCandidateAdmissible { candidate: _, fit: _ } => false @@ -686,6 +690,7 @@ fn is_semantic_gap(missing: MissingFact) -> Bool { MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false MemoryAttemptFailedForNonMemoryCause { cause: _ } => false + EvidenceIdentityIncomparable { axis: _ } => false PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false } } @@ -747,6 +752,7 @@ test fn w_only_a_typed_memory_refusal_rejects_on_memory() -> Bool { match m { MemoryAttemptFailedForNonMemoryCause { cause: c } => (c as String) == "chat template rendering failed" + EvidenceIdentityIncomparable { axis: _ } => false MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false @@ -772,6 +778,7 @@ fn is_contradiction_refusal(verdict: CandidateVerdict) -> Bool { token_count_value(t: f) == 400000 && n == 1 MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false MemoryAttemptFailedForNonMemoryCause { cause: _ } => false + EvidenceIdentityIncomparable { axis: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false } @@ -866,6 +873,7 @@ fn is_unobserved_verdict(verdict: CandidateVerdict) -> Bool { MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => true MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false MemoryAttemptFailedForNonMemoryCause { cause: _ } => false + EvidenceIdentityIncomparable { axis: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false } @@ -926,6 +934,96 @@ fn quiet_rate_at(node: NonEmptyStr, hot: Nat) -> TokensPerSecond? { ) } +// ================= AN UNANSWERABLE COMPARISON IS NOT A DIFFERENT REALIZATION ================= +// +// THE DEFECT, and it was committed one line under a comment warning against it: cross-family digest +// pairs were collapsed to `false`. An sha256 reading and an fnv1a reading of the SAME artifact then +// compared "different", the evidence silently left the population, and the candidate came back +// UNOBSERVED -- which reads as "nobody measured this" when the truth is "a measurement is held and +// cannot be lined up". Unknown is not rejected, and unanswerable is not unobserved. +// +// The fixture below is exactly that pair: identical release, quantization, runtime and resolved +// configuration; the artifact digest differs only in HASH FAMILY. +fn structural_digest(hex: String) -> ContentHash { + match content_hash_from_structural_digest(digest: hex) { + Present { value: h } => h + Absent => content_hash_of_value(value: "malformed structural digest" as NonEmptyStr) + } +} + +fn r_iq2_xxs_structural_artifact() -> ServingRealizationIdentity { + realization( + release_id: r_iq2_xxs.release_id, + quant: r_iq2_xxs.quant_label, + artifact: structural_digest(hex: "32af248f4cab44ff"), + runtime: ollama_runtime, + config: serial_config, + ) +} + +test fn w_a_cross_family_digest_refuses_rather_than_reading_as_unobserved() -> Bool { + let cross_family = with_memory_observations(base: build_iq2_xxs, observations: [ + served(r: r_iq2_xxs_structural_artifact(), node: serving_node, instrument: ps_instrument, + depth: 400000, sessions: 1, resident: 87692389907), + ]) + match evaluate_candidate(candidate: cross_family, node: serving_node, constraints: floor_400k) { + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => + match m { + EvidenceIdentityIncomparable { axis: a } => (a as String) == "artifact digest" + MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => false + MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false + MemoryAttemptFailedForNonMemoryCause { cause: _ } => false + SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false + PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false + } + ServingCandidateAdmissible { candidate: _, fit: _ } => false + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false + } +} + +// THE THREE STATES ARE ALL REACHABLE, and the ordering rule holds: a definite mismatch on any axis +// SETTLES the question and beats an undecided one, because these are plainly not the same +// realization whatever a second axis could not decide. Only when every other axis matches does the +// undecided one become load-bearing and block the answer. +test fn w_realization_comparison_reaches_all_three_states_and_different_wins() -> Bool { + let cross_family = r_iq2_xxs_structural_artifact() + let other_release_cross_family = realization( + release_id: qwen_release, + quant: r_iq2_xxs.quant_label, + artifact: structural_digest(hex: "32af248f4cab44ff"), + runtime: ollama_runtime, + config: serial_config, + ) + is_same(c: compare_serving_realization(a: r_iq2_xxs, b: r_iq2_xxs)) + && is_different(c: compare_serving_realization(a: r_iq2_xxs, b: r_qwen)) + && is_incomparable(c: compare_serving_realization(a: r_iq2_xxs, b: cross_family)) + && is_different(c: compare_serving_realization(a: r_iq2_xxs, b: other_release_cross_family)) +} + +fn is_same(c: RealizationComparison) -> Bool { + match c { + RealizationSame => true + RealizationDifferent => false + RealizationIncomparable { axis: _ } => false + } +} + +fn is_different(c: RealizationComparison) -> Bool { + match c { + RealizationSame => false + RealizationDifferent => true + RealizationIncomparable { axis: _ } => false + } +} + +fn is_incomparable(c: RealizationComparison) -> Bool { + match c { + RealizationSame => false + RealizationDifferent => false + RealizationIncomparable { axis: _ } => true + } +} + test fn w_all_serving_choice_claims_hold() -> Bool { w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolved() && w_the_real_roster_refuses_while_the_higher_rank_is_unresolved() @@ -947,6 +1045,8 @@ test fn w_all_serving_choice_claims_hold() -> Bool { && w_a_refusal_generalizes_to_no_easier_point_but_a_success_does() && w_the_concurrency_axis_carries_the_same_asymmetry() && w_a_prefill_reading_at_lower_concurrency_or_another_node_does_not_qualify() + && w_a_cross_family_digest_refuses_rather_than_reading_as_unobserved() + && w_realization_comparison_reaches_all_three_states_and_different_wins() } // ======================= THE ANSWERABILITY CLAIMS ======================= @@ -974,6 +1074,7 @@ test fn w_an_unmeasured_candidate_is_unanswerable_not_rejected() -> Bool { token_count_value(t: f) == 400000 && n == 1 MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => false MemoryAttemptFailedForNonMemoryCause { cause: _ } => false + EvidenceIdentityIncomparable { axis: _ } => false PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => false SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => false } From 022c34ab328b7cd4405a43e55b298d335790fcb2 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Wed, 2 Sep 2026 01:36:35 +0000 Subject: [PATCH 17/22] An incomparable axis stalls only where the observation would have counted Five blockers, and one of them was a live defect rather than a hardening. INCOMPARABILITY SCOPE. first_incomparable_axis filtered candidate evidence by node alone before asking whether identities compared, so an observation that could never have participated still stalled the candidate. A footprint at 8,192 tokens is irrelevant to a 400,000-token demand on depth alone; recorded in a different hash family it made the whole candidate unanswerable. Every extra reading on a node made that node LESS answerable, which inverts what evidence is for. An observation now clears its point conditions -- the asymmetric ones already established here, a success generalizing and a refusal speaking only at its exact point -- before its identity is consulted at all. MALFORMED DIGESTS ARE UNWRITABLE, not fallen back on. The witness mapped an unparseable hex literal to a shared structural hash, so every typo compared EQUAL to every other typo. Sha256DigestHex is a String where lower_hex_64 and the substrate checks it at typecheck for a literal -- measured: "zz" as Sha256DigestHex is a resolve-time mismatch. The helper is total with no fallback arm. Rung 4, not a repaired rung 2. PREFILL BINDS EXACT CONCURRENCY. The old rule qualified a busier reading for a quieter demand on the sentence "both axes cost throughput" -- an assertion about a scheduler, not a fact about the work. If rate is server-aggregate rather than per-request, continuous batching points it the other way, and nothing on the carrier declares which. Depth keeps its direction because prefill re-reads the context; concurrency loses it until a metric-semantics and monotonicity receipt exists. DETERMINISTIC FIT SELECTION. shallowest compared depth alone, so two qualifying readings at one depth tied and resolved by roster order -- the cited witness changed with the input sequence. Total lexicographic order over depth, sessions, footprint, instrument. CITED VERSUS OBSERVED RUNTIME IDENTITY. A release row is a closed authority, so its identity is total; a reading off a live host is exactly the open question, so its identity is optional. Both route through one seed. The all-rejected witness asserted length(r) == 2, which any function emitting two rows satisfied. It is now an identity join over quantization label with the memory arm and observed refusal checked. Every new witness verified discriminating by reverting the rule under test: the busier-prefill, irrelevant-cross-family and mixed-receipt probes each go red against the previous behavior and green against this one. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01U397y4s3dSBof7vGPAX87G --- dag/gunbc/model/choice.dag | 302 ++++++++++++++---- .../model/serving_choice_witness_test.dag | 279 ++++++++++++---- 2 files changed, 470 insertions(+), 111 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index edacfdc4264..cfae9c7d974 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -6,7 +6,15 @@ import gunbc.model.publication { ReleaseIdentity, release_identity_equal } import std.content_hash { ContentHash, ContentHashComparison, ContentHashEqual, ContentHashDifferent, ContentHashCrossFamilyIncomparable, compare_content_hash, + Fnv1a64Structural, content_hash_atom, content_hash_eq_structural, + content_hash_combine_structural, } +import gunbc.ollama_runtime_bundle { ollama_runtime_materialization_identity } +import extdeps.ollama.binary_release { + OllamaBinaryRelease, ollama_binary_release_for_observed_asset_digest, + ollama_reported_version_matches_release, +} +import product.placement_supply { HostIdentity, host_identity_eq } import std.measure { ByteSize, byte_size, byte_size_count, TokenCount, token_count, token_count_value, @@ -97,25 +105,121 @@ import std.measure { // observes today. It is a sole_constructor carrier with one mint, so a bare string cannot travel as // a runtime identity. NEXT-RUNG TRIGGER for this axis alone: an observed runtime build identity -- // the binary's own digest or a version endpoint reading -- at which point it joins the two above. +// SEALING A CARRIER DOES NOT GROUND ITS INPUTS, which is what the predecessor got wrong here. +// ServingRuntimeIdentity was sole_constructor -- and its one mint accepted a freely authored +// {name, version}, so two different Ollama binaries both minted `ollama x 0.32.9` and compared +// equal. The seal stopped an outside record literal and did nothing about the fiction inside it. +// +// AND THE REPOSITORY ALREADY OWNS THIS FACT. gunbc.ollama_runtime_bundle +// ollama_runtime_materialization_identity_for_observed_release maps an observed /api/version wire +// PLUS a cited release-asset digest to exactly one catalog row, and refuses when they disagree -- +// written precisely because the version wire alone is ambiguous across the generic-cuda and jetpack +// builds at v0.32.9. A second {name, version} carrier beside it was a nickname for that authority +// (section 3), so it is deleted rather than repaired. +// +// The only mint now delegates there. A runtime identity therefore cannot exist without an observed +// version and an observed asset digest that agree on one catalog row. type ServingRuntimeIdentity sole_constructor { - name: NonEmptyStr - version: NonEmptyStr + materialization: Fnv1a64Structural +} + +// TWO MINTS, AND THE DIFFERENCE IS EPISTEMIC, NOT CONVENIENCE. A CITED release row is a closed +// authority -- the bytes, arch, variant, codec and entrypoint are all in hand -- so its identity is +// TOTAL. An OBSERVED runtime is a reading off a live host, and whether that reading names a release +// we know is exactly the question, so its identity is OPTIONAL. Both route through the same seed, so +// an observation that does name a cited release mints the identity that release already had. +fn serving_runtime_identity_for_release(rel: OllamaBinaryRelease) -> ServingRuntimeIdentity { + ServingRuntimeIdentity { materialization: ollama_runtime_materialization_identity(rel: rel) } } -fn serving_runtime_identity(name: NonEmptyStr, version: NonEmptyStr) -> ServingRuntimeIdentity { - ServingRuntimeIdentity { name: name, version: version } +fn serving_runtime_identity_from_observation( + version: String, + asset_digest_hex: String, +) -> ServingRuntimeIdentity? { + match ollama_binary_release_for_observed_asset_digest(digest_hex: asset_digest_hex) { + Absent => none + Present { value: rel } => + if ollama_reported_version_matches_release(version: version, rel: rel) { + Present { value: serving_runtime_identity_for_release(rel: rel) } + } else { + none + } + } } fn serving_runtime_equal(a: ServingRuntimeIdentity, b: ServingRuntimeIdentity) -> Bool { - (a.name as String) == (b.name as String) && (a.version as String) == (b.version as String) + content_hash_eq_structural(left: a.materialization, right: b.materialization) +} + +// TWO CONFIGURATION GRAINS, because one digest cannot answer for both and conflating them was a +// measured inconsistency in this module's own fixture: a single argv digest was carried for readings +// at 131,072, 262,144, 400,000 and 1,048,576 tokens, while the resolved argv encodes the context and +// the slot count through `-c` and `-np`. One exact full-argv digest cannot identify four points. +// +// FIXED MODE is the argv with the VARIED axes removed -- the flags held constant across a +// measurement series (`--flash`, `--load-mode`, the model path). It is what two readings must share +// before one can bound the other, and it is DERIVED from the observed argv rather than authored. +// +// EXACT CONFIGURATION is the whole argv of ONE measured point. It belongs on the observation, not on +// the realization, because it changes per reading by construction. +// +// Both are derived by the functions below from an argv the observer actually read. There is no mint +// that accepts a bare hash: a digest with no argv behind it is the claim-shaped input this repair +// exists to remove. +type FixedRuntimeModeIdentity sole_constructor { + mode_digest: Fnv1a64Structural +} + +fn argv_digest_of(argv: List) -> Fnv1a64Structural { + fold(argv, content_hash_atom(value: "argv" as NonEmptyStr), (acc, word) => + content_hash_combine_structural(left: acc, right: content_hash_atom(value: word))) } -type ResolvedRuntimeConfiguration sole_constructor { - argv_digest: ContentHash +// The varied flags AND their values are DROPPED, not blanked, so the mode digest is invariant to +// their values by construction rather than by the caller remembering to pass matching ones. +// +// A LEFT SCAN, NOT AN INDEX LOOKUP. The predecessor walked positions and read argv[i-1] through an +// at_index helper whose out-of-range arm returned `"" as NonEmptyStr` -- a value that is not a +// NonEmptyStr at all, fabricated to keep a total signature. The scan carries the one bit that +// lookup was reconstructing: whether the PREVIOUS word was a varied flag, so its value is the word +// to drop. Nothing can go out of range, so there is no arm to fabricate for. +type ArgvScan { + kept: List + drop_next: Bool +} + +fn argv_scan_step(acc: ArgvScan, word: NonEmptyStr, varied: List) -> ArgvScan { + if acc.drop_next { + ArgvScan { kept: acc.kept, drop_next: false } + } else { + if word_in(roster: varied, word: word) { + ArgvScan { kept: acc.kept, drop_next: true } + } else { + ArgvScan { kept: flat_map([acc.kept, [word]], g => g), drop_next: false } + } + } } -fn resolved_runtime_configuration(argv_digest: ContentHash) -> ResolvedRuntimeConfiguration { - ResolvedRuntimeConfiguration { argv_digest: argv_digest } +fn drop_varied_flags(argv: List, varied: List) -> List { + fold(argv, ArgvScan { kept: [], drop_next: false }, (acc, word) => + argv_scan_step(acc: acc, word: word, varied: varied)).kept +} + +fn word_in(roster: List, word: NonEmptyStr) -> Bool { + length(filter(roster, r => (r as String) == (word as String))) > 0 +} + +fn fixed_runtime_mode_from_argv( + argv: List, + varied_flags: List, +) -> FixedRuntimeModeIdentity { + FixedRuntimeModeIdentity { + mode_digest: argv_digest_of(argv: drop_varied_flags(argv: argv, varied: varied_flags)), + } +} + +fn fixed_runtime_mode_equal(a: FixedRuntimeModeIdentity, b: FixedRuntimeModeIdentity) -> Bool { + content_hash_eq_structural(left: a.mode_digest, right: b.mode_digest) } type ServingRealizationIdentity sole_constructor { @@ -123,7 +227,7 @@ type ServingRealizationIdentity sole_constructor { quant_label: NonEmptyStr artifact_digest: ContentHash runtime: ServingRuntimeIdentity - runtime_config: ResolvedRuntimeConfiguration + fixed_mode: FixedRuntimeModeIdentity } // The only mint. sole_constructor already refuses a record literal outside this module; this is the @@ -133,14 +237,14 @@ fn serving_realization_identity( quant_label: NonEmptyStr, artifact_digest: ContentHash, runtime: ServingRuntimeIdentity, - runtime_config: ResolvedRuntimeConfiguration, + fixed_mode: FixedRuntimeModeIdentity, ) -> ServingRealizationIdentity { ServingRealizationIdentity { release_id: release_id, quant_label: quant_label, artifact_digest: artifact_digest, runtime: runtime, - runtime_config: runtime_config, + fixed_mode: fixed_mode, } } @@ -212,9 +316,7 @@ fn compare_serving_realization( ), later: realization_comparison_then( earlier: compare_bool_axis(same: serving_runtime_equal(a: a.runtime, b: b.runtime)), - later: compare_digest_axis( - axis: "resolved runtime configuration digest" as NonEmptyStr, - a: a.runtime_config.argv_digest, b: b.runtime_config.argv_digest), + later: compare_bool_axis(same: fixed_runtime_mode_equal(a: a.fixed_mode, b: b.fixed_mode)), ), ) } @@ -316,7 +418,7 @@ fn serving_regime_wire(regime: ServingRegime) -> String { // on one node does not describe another. Both now join. type FreshPrefillObservation { realization: ServingRealizationIdentity - node: NonEmptyStr + node: HostIdentity concurrent_sessions: Nat depth: TokenCount rate: TokensPerSecond @@ -335,7 +437,7 @@ type FreshPrefillObservation { fn prefill_rate_at_floor( observations: List, realization: ServingRealizationIdentity, - node: NonEmptyStr, + node: HostIdentity, hot_sessions: Nat, floor: TokenCount, regime: ServingRegime, @@ -350,21 +452,41 @@ fn prefill_rate_at_floor( } } -// DEPTH AND CONCURRENCY QUALIFY IN THE SAME DIRECTION, and it is the conservative one: a rate -// measured DEEPER and under MORE concurrent load bounds the rate at a shallower, quieter demand, -// because both axes cost throughput. A shallower or quieter reading flatters the candidate, so it -// does not qualify. The node must match exactly -- it is not an axis with a direction. +// DEPTH CARRIES A DIRECTION. CONCURRENCY DOES NOT -- NOT YET, AND THE DIFFERENCE IS EVIDENCE. +// +// Depth qualifies conservatively: a rate measured DEEPER bounds the rate at a shallower demand, +// because prefill re-reads the whole context and attention cost grows with it, so a deep reading +// cannot flatter a shallow demand. That direction is grounded in what the work IS, not in how a +// particular server schedules it. +// +// The predecessor gave concurrency the same treatment on the same sentence -- "both axes cost +// throughput" -- and that is an ASSERTION ABOUT A SCHEDULER, not a fact about the work. It has two +// holes and either one inverts it. First, metric semantics: if `rate` is per-request tokens/s then +// more concurrency does lower it, but if it is the server's AGGREGATE prefill rate then continuous +// batching RAISES it with concurrency, and the same number read the other way makes a 4-session +// reading a flattering one rather than a bounding one. Nothing in FreshPrefillObservation declares +// which it is. Second, even granting per-request semantics, monotonicity across the batching +// threshold is unmeasured on this fleet -- a request that fits one slot and a request that waits in +// a queue are not two points on one curve. +// +// So concurrency binds EXACT until both receipts exist: a declared metric semantics on the carrier, +// and a measured monotonicity across the range the selector would interpolate over. Exact matching +// shrinks the qualifying population, so more candidates come back PrefillRateUnmeasuredAtFloor -- +// unanswerable rather than answered from a reading that may point the wrong way. That is the +// fail-closed direction, and the cost of it is visible in the refusal rather than hidden in a rate. +// +// The node must match exactly for a third reason: it is not an axis with a direction at all. fn slowest_fresh_rate_at_floor( observations: List, realization: ServingRealizationIdentity, - node: NonEmptyStr, + node: HostIdentity, hot_sessions: Nat, floor: TokenCount, ) -> TokensPerSecond? { let deep = filter(observations, o => serving_realization_equal(a: o.realization, b: realization) - && (o.node as String) == (node as String) - && o.concurrent_sessions >= hot_sessions + && host_identity_eq(a: o.node, b: node) + && o.concurrent_sessions == hot_sessions && token_count_value(t: o.depth) >= token_count_value(t: floor)) fold(deep, none, (worst, o) => match worst { @@ -432,7 +554,7 @@ type RunnerAttemptOutcome type OllamaRunnerMemoryObservation { realization: ServingRealizationIdentity - node: NonEmptyStr + node: HostIdentity instrument: NonEmptyStr context_depth: TokenCount concurrent_sessions: Nat @@ -478,30 +600,16 @@ type MemoryFitEvidence // it is not counted as a different realization: skipping it would silently shrink the population the // verdict is drawn from, which is how a held measurement turns into "unobserved". The axis that // could not be lined up is carried out so the refusal names what to fix. -fn first_incomparable_axis( - observations: List, - realization: ServingRealizationIdentity, - node: NonEmptyStr, -) -> NonEmptyStr? { - let on_this_node = filter(observations, o => (o.node as String) == (node as String)) - fold(on_this_node, none, (found, o) => - match found { - Present { value: a } => Present { value: a } - Absent => incomparable_axis( - comparison: compare_serving_realization(a: o.realization, b: realization)) - }) -} - fn served_qualifying_observations( observations: List, realization: ServingRealizationIdentity, floor: TokenCount, hot_sessions: Nat, - node: NonEmptyStr, + node: HostIdentity, ) -> List { filter(observations, o => serving_realization_equal(a: o.realization, b: realization) - && (o.node as String) == (node as String) + && host_identity_eq(a: o.node, b: node) && token_count_value(t: o.context_depth) >= token_count_value(t: floor) && o.concurrent_sessions >= hot_sessions && outcome_is_served(o: o)) @@ -512,11 +620,11 @@ fn refused_at_exact_point( realization: ServingRealizationIdentity, floor: TokenCount, hot_sessions: Nat, - node: NonEmptyStr, + node: HostIdentity, ) -> List { filter(observations, o => serving_realization_equal(a: o.realization, b: realization) - && (o.node as String) == (node as String) + && host_identity_eq(a: o.node, b: node) && token_count_value(t: o.context_depth) == token_count_value(t: floor) && o.concurrent_sessions == hot_sessions && outcome_is_memory_refusal(o: o)) @@ -529,11 +637,11 @@ fn failed_otherwise_at_exact_point( realization: ServingRealizationIdentity, floor: TokenCount, hot_sessions: Nat, - node: NonEmptyStr, + node: HostIdentity, ) -> List { filter(observations, o => serving_realization_equal(a: o.realization, b: realization) - && (o.node as String) == (node as String) + && host_identity_eq(a: o.node, b: node) && token_count_value(t: o.context_depth) == token_count_value(t: floor) && o.concurrent_sessions == hot_sessions && outcome_is_other_failure(o: o)) @@ -576,17 +684,102 @@ fn outcome_is_other_failure(o: OllamaRunnerMemoryObservation) -> Bool { served: false, refused_for_memory: false, failed_otherwise: true) } +// INCOMPARABILITY IS ONLY LOAD-BEARING WHERE THE OBSERVATION WOULD OTHERWISE HAVE PARTICIPATED. +// +// The predecessor filtered on node alone and asked every remaining observation whether its identity +// compared. That makes an IRRELEVANT reading load-bearing: a footprint taken at 8,192 tokens whose +// artifact digest was recorded in a different hash family has nothing to say about a 400,000-token +// demand, but under a node-only filter its undecidable axis stalls the candidate outright. The whole +// roster then becomes a stall surface, and the more evidence a node accumulates the less answerable +// it gets -- precisely backwards. +// +// The rule is that an observation must clear its POINT conditions before its identity matters. Those +// conditions are independent of identity, so they can be evaluated on an observation whose identity +// cannot be compared, and they are the asymmetric ones already established elsewhere in this module: +// a success qualifies from a deeper and busier point, a refusal or a non-memory failure speaks only +// at its exact point. An observation that fails them is irrelevant on facts that have nothing to do +// with the digest, and stays irrelevant. +fn participates_at_point( + o: OllamaRunnerMemoryObservation, + floor: TokenCount, + hot_sessions: Nat, +) -> Bool { + let generalizes = token_count_value(t: o.context_depth) >= token_count_value(t: floor) + && o.concurrent_sessions >= hot_sessions + let exact = token_count_value(t: o.context_depth) == token_count_value(t: floor) + && o.concurrent_sessions == hot_sessions + runner_attempt_fold(outcome: o.outcome, + served: generalizes, refused_for_memory: exact, failed_otherwise: exact) +} + +fn first_incomparable_axis( + observations: List, + realization: ServingRealizationIdentity, + floor: TokenCount, + hot_sessions: Nat, + node: HostIdentity, +) -> NonEmptyStr? { + let relevant = filter(observations, o => + host_identity_eq(a: o.node, b: node) + && participates_at_point(o: o, floor: floor, hot_sessions: hot_sessions)) + fold(relevant, none, (found, o) => + match found { + Present { value: a } => Present { value: a } + Absent => incomparable_axis( + comparison: compare_serving_realization(a: o.realization, b: realization)) + }) +} + // The SHALLOWEST qualifying attempt, because it is the one closest to the configuration actually // requested. Depth is the axis the demand is expressed in, so ordering on it is ordering on the // question rather than on an incidental byte count. +// +// ORDER-INDEPENDENCE IS THE WHOLE POINT, AND DEPTH ALONE DOES NOT BUY IT. The qualifying population +// is filtered by depth >= floor and sessions >= hot_sessions, so two readings can qualify at the +// SAME depth and differ on concurrency, on reported footprint, or on the instrument that produced +// them. Comparing on depth alone leaves those tied, and a tie resolved by taking the incumbent is +// resolved by ROSTER ORDER: the same evidence set handed to the selector in a different sequence +// reports a different witness, and the receipt a refusal cites stops being a function of the facts. +// That is the same defect this module refuses elsewhere -- an answer that depends on how the +// question was spelled -- so the comparison is a TOTAL lexicographic order over every axis a +// qualifying pair can differ on. The last key is the instrument string, which is what finally +// separates two otherwise identical readings; identical on all four and the observations are +// interchangeable, so which one is returned carries no information. +fn observation_order_key_before( + a: OllamaRunnerMemoryObservation, + b: OllamaRunnerMemoryObservation, +) -> Bool { + let ad = token_count_value(t: a.context_depth) + let bd = token_count_value(t: b.context_depth) + if ad != bd { ad < bd } else { + if a.concurrent_sessions != b.concurrent_sessions { + a.concurrent_sessions < b.concurrent_sessions + } else { + let ab = reported_total_bytes(o: a) + let bb = reported_total_bytes(o: b) + if ab != bb { ab < bb } else { + (a.instrument as String) < (b.instrument as String) + } + } + } +} + +// The footprint axis is only populated for a SERVED attempt; a refusal reports no buffer. Both +// non-served arms sort as zero, which is not a fabricated measurement -- the three arms never share +// a comparison, because memory_fit_evidence partitions by outcome before it orders. +fn reported_total_bytes(o: OllamaRunnerMemoryObservation) -> Int { + match o.outcome { + RunnerServed { reported_total_buffer: t, reported_gpu_buffer: _ } => byte_size_count(b: t) + RunnerRefusedForMemory { detail: _ } => 0 + RunnerFailedForOtherCause { cause: _ } => 0 + } +} + fn shallowest( a: OllamaRunnerMemoryObservation, b: OllamaRunnerMemoryObservation, ) -> OllamaRunnerMemoryObservation { - match token_count_value(t: a.context_depth) < token_count_value(t: b.context_depth) { - true => a - false => b - } + if observation_order_key_before(a: a, b: b) { a } else { b } } fn shallowest_of( @@ -605,10 +798,11 @@ fn memory_fit_evidence( realization: ServingRealizationIdentity, floor: TokenCount, hot_sessions: Nat, - node: NonEmptyStr, + node: HostIdentity, ) -> MemoryFitEvidence { match first_incomparable_axis( - observations: observations, realization: realization, node: node) { + observations: observations, realization: realization, + floor: floor, hot_sessions: hot_sessions, node: node) { Present { value: axis } => MemoryFitIdentityIncomparable { axis: axis } Absent => decidable_memory_fit_evidence( observations: observations, realization: realization, @@ -621,7 +815,7 @@ fn decidable_memory_fit_evidence( realization: ServingRealizationIdentity, floor: TokenCount, hot_sessions: Nat, - node: NonEmptyStr, + node: HostIdentity, ) -> MemoryFitEvidence { let served = served_qualifying_observations( observations: observations, realization: realization, @@ -787,7 +981,7 @@ fn rejected(candidate: QuantizedCandidate, axis: RejectionAxis) -> CandidateVerd fn evaluate_candidate( candidate: QuantizedCandidate, - node: NonEmptyStr, + node: HostIdentity, constraints: ServingConstraints, ) -> CandidateVerdict { match memory_fit_evidence( @@ -972,7 +1166,7 @@ fn unanswerable_cause_wire(cause: UnanswerableCause) -> String { fn choose_serving_candidate( candidates: List, - node: NonEmptyStr, + node: HostIdentity, constraints: ServingConstraints, ) -> ServingChoice { let verdicts = map(candidates, c => diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index 725a1a5ccf4..1a91dfdb004 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -9,15 +9,19 @@ import std.measure { } import gunbc.model.publication { ReleaseIdentity } import std.content_hash { - ContentHash, sha256_hex_digest, as_content_hash_cryptographic, content_hash_of_value, - content_hash_from_structural_digest, + ContentHash, Sha256Digest, Sha256DigestHex, as_content_hash_cryptographic, + Fnv1a64StructuralDigestHex, structural_content_hash, as_content_hash_structural, +} +import extdeps.ollama.binary_release { + ollama_v0_32_9_release, ollama_v0_32_9_jetpack5_release, ollama_v0_32_9_jetpack6_release, } import gunbc.model.choice { PrefillRateUnmeasuredAtFloor, SemanticContextUnverifiedAtFloor, FreshPrefillObservation, prefill_rate_at_floor, ServingRealizationIdentity, serving_realization_equal, serving_realization_identity, - ServingRuntimeIdentity, serving_runtime_identity, - ResolvedRuntimeConfiguration, resolved_runtime_configuration, + ServingRuntimeIdentity, serving_runtime_identity_for_release, + serving_runtime_identity_from_observation, + FixedRuntimeModeIdentity, fixed_runtime_mode_from_argv, SemanticContextEvidence, OllamaRunnerMemoryObservation, memory_fit_evidence, RunnerAttemptOutcome, RunnerServed, RunnerRefusedForMemory, RunnerFailedForOtherCause, @@ -25,6 +29,7 @@ import gunbc.model.choice { MemoryFitAttemptFailedOtherwise, MemoryFitUnobserved, MemoryFitContradictedAtConfiguration, MemoryAttemptFailedForNonMemoryCause, EvidenceIdentityIncomparable, MemoryFitIdentityIncomparable, + outcome_is_memory_refusal, DoesNotFitMemory, ContextBelowFloor, PrefillBelowFloor, RealizationComparison, RealizationSame, RealizationDifferent, RealizationIncomparable, compare_serving_realization, FreshSessionPrefill, WarmContinuation, serving_regime_wire, @@ -71,28 +76,42 @@ data fixture_node: NonEmptyStr = "fixture-node" as NonEmptyStr // `size_vram` agreed exactly on this unified-memory node, so nothing turns on which field is read. data ps_instrument: NonEmptyStr = "/api/ps size and size_vram after load at explicit num_ctx" as NonEmptyStr -// THE RESOLVED ARGV, hashed -- not the desired environment. The readings below were taken while the -// runner had been launched with OLLAMA_NUM_PARALLEL unset, which resolved to a single slot; the -// digest stands for that exact argv. The fleet has since moved to four slots, which is a DIFFERENT -// configuration and therefore a different key, so these readings correctly stop answering for it. -data serial_config: ResolvedRuntimeConfiguration = resolved_runtime_configuration( - argv_digest: digest(hex: "1c3f5d7981b3d5f719a2c4e60815273949e1b3d5f7092a4c6e80b2d4f6183a5c"), +// THE FIXED RUNTIME MODE, hashed from the resolved argv with the VARIED flags dropped. The readings +// below were taken with OLLAMA_NUM_PARALLEL unset and num_ctx set per reading; those two are the +// axes the selector varies and asks about, so hashing their values into the mode key would make +// every reading answer for exactly one point and nothing else. They are dropped, not blanked, so the +// key is invariant to them by construction. Everything else in the argv IS in the key: a runner +// launched with different flash-attention or KV-cache-type settings is a different mode, and its +// footprint readings correctly stop answering for this one. +data varied_flags: List = [ + "--num-ctx" as NonEmptyStr, + "--parallel" as NonEmptyStr, +] + +data serial_mode: FixedRuntimeModeIdentity = fixed_runtime_mode_from_argv( + argv: [ + "ollama" as NonEmptyStr, "serve" as NonEmptyStr, + "--num-ctx" as NonEmptyStr, "400000" as NonEmptyStr, + "--parallel" as NonEmptyStr, "1" as NonEmptyStr, + ], + varied_flags: varied_flags, ) -data ollama_runtime: ServingRuntimeIdentity = serving_runtime_identity( - name: "ollama" as NonEmptyStr, - version: "0.32.9" as NonEmptyStr, -) +// THE RUNTIME, from the CITED pinned release rather than a name-and-version pair the witness types. +// `serving_runtime_identity_for_release` is total because the release row is a closed authority; the +// observation-side mint beside it is the one that can come back Absent, and w_an_unknown_runtime_ +// asset_has_no_identity below holds it to that. +data ollama_runtime: ServingRuntimeIdentity = serving_runtime_identity_for_release(rel: ollama_v0_32_9_release) -// A DIGEST, NOT A TAG. Every hex string below is 64 lower-hex characters and is admitted through -// std.content_hash's validated constructor, so a malformed digest has no value here rather than a -// failing check. The IQ2_XXS one is the digest /api/ps reports for the loaded build on the fleet; -// re-derive it from that endpoint rather than reading it here. -fn digest(hex: String) -> ContentHash { - match sha256_hex_digest(hex: hex) { - Present { value: d } => as_content_hash_cryptographic(digest: d) - Absent => content_hash_of_value(value: "malformed digest literal" as NonEmptyStr) - } +// A DIGEST, NOT A TAG, AND A MALFORMED ONE IS UNWRITABLE. Every hex string below is admitted as +// Sha256DigestHex, a `String where lower_hex_64` refinement the substrate checks at TYPECHECK for a +// literal -- measured: `"zz" as Sha256DigestHex` is a resolve-time type mismatch, not a runtime +// none. So this helper is total with no fallback arm, and the predecessor's shared +// "malformed digest literal" structural hash -- which made every typo compare EQUAL to every other +// typo -- is gone rather than repaired. The IQ2_XXS digest is what /api/ps reports for the loaded +// build on the fleet; re-derive it from that endpoint rather than reading it here. +fn digest(hex: Sha256DigestHex) -> ContentHash { + as_content_hash_cryptographic(digest: Sha256Digest { hex: hex }) } fn realization( @@ -100,14 +119,14 @@ fn realization( quant: NonEmptyStr, artifact: ContentHash, runtime: ServingRuntimeIdentity, - config: ResolvedRuntimeConfiguration, + fixed_mode: FixedRuntimeModeIdentity, ) -> ServingRealizationIdentity { serving_realization_identity( release_id: release_id, quant_label: quant, artifact_digest: artifact, runtime: runtime, - runtime_config: config, + fixed_mode: fixed_mode, ) } @@ -162,44 +181,44 @@ data qwen_release: ReleaseIdentity = ReleaseIdentity { data r_iq2_xxs: ServingRealizationIdentity = realization( release_id: deepseek_v4_flash_release, quant: "IQ2_XXS" as NonEmptyStr, - artifact: digest(hex: "32af248f4cab44ffcb631298f66ff0bdf92a9350a3763a86b9a5b304fc4c1e64"), + artifact: digest(hex: "32af248f4cab44ffcb631298f66ff0bdf92a9350a3763a86b9a5b304fc4c1e64" as Sha256DigestHex), runtime: ollama_runtime, - config: serial_config, + fixed_mode: serial_mode, ) data r_iq3_s: ServingRealizationIdentity = realization( release_id: deepseek_v4_flash_release, quant: "UD-IQ3_S" as NonEmptyStr, - artifact: digest(hex: "5b7e1042cc9d3fa6081b4e2d7c3950ae61f8d24b0937ce5182a4f6b0d3e79c18"), + artifact: digest(hex: "5b7e1042cc9d3fa6081b4e2d7c3950ae61f8d24b0937ce5182a4f6b0d3e79c18" as Sha256DigestHex), runtime: ollama_runtime, - config: serial_config, + fixed_mode: serial_mode, ) data r_unmeasured: ServingRealizationIdentity = realization( release_id: deepseek_v4_flash_release, quant: "UD-IQ4_XS" as NonEmptyStr, - artifact: digest(hex: "9d2c48b6103fae57c81d0b2e64937fa5081c3d6e9b204f7a15c8e0b3d629f47a"), + artifact: digest(hex: "9d2c48b6103fae57c81d0b2e64937fa5081c3d6e9b204f7a15c8e0b3d629f47a" as Sha256DigestHex), runtime: ollama_runtime, - config: serial_config, + fixed_mode: serial_mode, ) data r_qwen: ServingRealizationIdentity = realization( release_id: qwen_release, quant: "Q8_0" as NonEmptyStr, - artifact: digest(hex: "a41f70c39b2d85e6014c7f3a92b60d58e7139fac402b6d81c53e97a0f42b18d6"), + artifact: digest(hex: "a41f70c39b2d85e6014c7f3a92b60d58e7139fac402b6d81c53e97a0f42b18d6" as Sha256DigestHex), runtime: ollama_runtime, - config: serial_config, + fixed_mode: serial_mode, ) fn fixture_realization(quant: NonEmptyStr) -> ServingRealizationIdentity { realization( release_id: fixture_release, quant: quant, - artifact: digest(hex: "ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00"), - runtime: serving_runtime_identity( - name: "fixture-runtime" as NonEmptyStr, version: "0" as NonEmptyStr), - config: resolved_runtime_configuration( - argv_digest: digest(hex: "beef00beef00beef00beef00beef00beef00beef00beef00beef00beef00beef")), + artifact: digest(hex: "ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00" as Sha256DigestHex), + runtime: serving_runtime_identity_for_release(rel: ollama_v0_32_9_jetpack6_release), + fixed_mode: fixed_runtime_mode_from_argv( + argv: ["fixture-runner" as NonEmptyStr, "--fixture" as NonEmptyStr], + varied_flags: varied_flags), ) } @@ -261,6 +280,15 @@ data floor_400k: ServingConstraints = ServingConstraints { prefill_regime: FreshSessionPrefill, } +// The same floor at four concurrent sessions, so one candidate can be asked the same memory +// question at two points on the concurrency axis. +data floor_400k_four_sessions: ServingConstraints = ServingConstraints { + context_floor: token_count(count: 400000), + hot_sessions: 4, + prefill_floor: tokens_per_second(count: 0), + prefill_regime: FreshSessionPrefill, +} + data floor_8k: ServingConstraints = ServingConstraints { context_floor: token_count(count: 8192), hot_sessions: 1, @@ -422,27 +450,29 @@ fn iq2_xxs_answers_for(r: ServingRealizationIdentity, node: NonEmptyStr) -> Bool test fn w_evidence_from_another_realization_or_node_cannot_qualify() -> Bool { let wrong_release = realization( release_id: qwen_release, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_digest, - runtime: r_iq2_xxs.runtime, config: r_iq2_xxs.runtime_config) + runtime: r_iq2_xxs.runtime, fixed_mode: r_iq2_xxs.fixed_mode) let wrong_quant = realization( release_id: r_iq2_xxs.release_id, quant: "IQ3_S" as NonEmptyStr, artifact: r_iq2_xxs.artifact_digest, - runtime: r_iq2_xxs.runtime, config: r_iq2_xxs.runtime_config) + runtime: r_iq2_xxs.runtime, fixed_mode: r_iq2_xxs.fixed_mode) let wrong_artifact = realization( release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, - artifact: digest(hex: "7e3b91d0a5c82f461937be04d2a86c50f18e2b7d940c36a1e582f4b0c7d91836"), - runtime: r_iq2_xxs.runtime, config: r_iq2_xxs.runtime_config) + artifact: digest(hex: "7e3b91d0a5c82f461937be04d2a86c50f18e2b7d940c36a1e582f4b0c7d91836" as Sha256DigestHex), + runtime: r_iq2_xxs.runtime, fixed_mode: r_iq2_xxs.fixed_mode) let wrong_runtime = realization( release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_digest, - runtime: serving_runtime_identity(name: "ollama" as NonEmptyStr, version: "0.33.0" as NonEmptyStr), config: r_iq2_xxs.runtime_config) - let wrong_config = realization( + runtime: serving_runtime_identity_for_release(rel: ollama_v0_32_9_jetpack5_release), + fixed_mode: r_iq2_xxs.fixed_mode) + let wrong_mode = realization( release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_digest, - runtime: r_iq2_xxs.runtime, config: resolved_runtime_configuration( - argv_digest: digest(hex: "4444444444444444444444444444444444444444444444444444444444444444"))) + runtime: r_iq2_xxs.runtime, fixed_mode: fixed_runtime_mode_from_argv( + argv: ["ollama" as NonEmptyStr, "serve" as NonEmptyStr, "--flash-attn" as NonEmptyStr], + varied_flags: varied_flags)) iq2_xxs_answers_for(r: r_iq2_xxs, node: serving_node) && !iq2_xxs_answers_for(r: wrong_release, node: serving_node) && !iq2_xxs_answers_for(r: wrong_quant, node: serving_node) && !iq2_xxs_answers_for(r: wrong_artifact, node: serving_node) && !iq2_xxs_answers_for(r: wrong_runtime, node: serving_node) - && !iq2_xxs_answers_for(r: wrong_config, node: serving_node) + && !iq2_xxs_answers_for(r: wrong_mode, node: serving_node) && !iq2_xxs_answers_for(r: r_iq2_xxs, node: fixture_node) } @@ -607,13 +637,43 @@ test fn w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() -> Bool // What is left is the only thing the instrument can actually report as a non-fit: a configuration // that was attempted on the node and DID NOT PRODUCE A TOKEN. That is an observation, not a // calculation, and it is the arm a real out-of-memory load would populate. +// COUNTING TWO REJECTIONS IS NOT CHECKING THEM. `length(r) == 2` was satisfied by any function that +// emitted two rows -- two copies of the same candidate, two rows on the wrong axis, or two +// unanswerable verdicts miscast as rejections -- which is the count-instead-of-identity-join that +// DESIGN's oracle rule names directly. The claim is an IDENTITY JOIN: both declared candidates are +// present, distinguished by quantization label, and each is rejected on the MEMORY axis carrying the +// refusal actually observed for it, rather than on a context or prefill axis that happens to fail +// too. The count survives as a closure check -- no third row -- not as the whole assertion. +fn rejection_on_memory_for( + rejections: List, + quant: NonEmptyStr, +) -> Bool { + length(filter(rejections, v => + match v { + ServingCandidateRejected { identity: _, quant_label: q, axis: a } => + (q as String) == (quant as String) + && match a { + DoesNotFitMemory { observed: o } => + token_count_value(t: o.context_depth) == 400000 + && outcome_is_memory_refusal(o: o) + ContextBelowFloor { declared: _, floor: _ } => false + PrefillBelowFloor { measured: _, floor: _ } => false + } + ServingCandidateAdmissible { candidate: _, fit: _ } => false + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => false + })) == 1 +} + test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { match choose_serving_candidate( candidates: [oom_low(), oom_high()], node: fixture_node, constraints: floor_400k, ) { - NoCandidateAdmissible { rejections: r } => length(r) == 2 + NoCandidateAdmissible { rejections: r } => + rejection_on_memory_for(rejections: r, quant: "LOW" as NonEmptyStr) + && rejection_on_memory_for(rejections: r, quant: "HIGH" as NonEmptyStr) + && length(r) == 2 ChoseCandidate { verdict: _ } => false SelectionUnanswerable { cause: _, evaluations: _ } => false } @@ -787,6 +847,56 @@ fn is_contradiction_refusal(verdict: CandidateVerdict) -> Bool { } } +// MIXED RECEIPTS IN ONE CANDIDATE, which is the case every witness above misses. Each of those +// hands the selector a SINGLE observation, so the served/refused/broken partition is only ever +// exercised with one list non-empty. The interesting failures live where two lists are populated at +// DIFFERENT points, because that is where an off-by-one in the qualification rule stops being +// visible: a rule that qualified refusals by `depth >= floor` instead of `depth == floor` passes +// every single-observation witness and fails only here. +fn fixture_served_at( + r: ServingRealizationIdentity, + depth: Nat, + sessions: Nat, + resident: Nat, +) -> OllamaRunnerMemoryObservation { + served(r: r, node: fixture_node, instrument: fixture_instrument, + depth: depth, sessions: sessions, resident: resident) +} + +// A SUCCESS AT THE FLOOR AND AN OOM ABOVE IT ARE NOT A CONTRADICTION. They are the ordinary shape of +// a real sweep -- it fits at 400k, it does not fit at 1M -- and the honest reading is that the 400k +// demand is ANSWERED and admissible. The deeper refusal speaks only at its own point. Widen the +// refusal filter to `depth >= floor` and this candidate flips to a contradiction refusal, which +// would make every complete sweep unanswerable at exactly the floor it measured. +test fn w_a_success_at_the_floor_survives_an_oom_above_it() -> Bool { + is_admissible(verdict: evaluate_candidate( + candidate: with_memory_observations(base: fixture_low_rank, observations: [ + fixture_served_at(r: r_fixture_low, depth: 400000, sessions: 1, resident: 28000000000), + fixture_attempt(r: r_fixture_low, depth: 1048576, outcome: out_of_memory), + ]), + node: fixture_node, constraints: floor_400k)) +} + +// THE SAME SHAPE ON THE CONCURRENCY AXIS, and it must answer DIFFERENTLY at the two demands rather +// than picking one story. One candidate holds a 1-session success and a 4-session OOM at the same +// depth. At a 1-session demand the success qualifies and the 4-session refusal is out of scope, so +// it is admissible. At a 4-session demand the success no longer qualifies -- it was measured +// quieter -- and the refusal is exactly on point, so it is a memory rejection. A selector that +// ignored the concurrency axis on either filter would have to give both demands the same verdict. +test fn w_one_candidate_answers_admissible_quiet_and_rejected_busy() -> Bool { + let mixed = with_memory_observations(base: fixture_low_rank, observations: [ + fixture_served_at(r: r_fixture_low, depth: 400000, sessions: 1, resident: 28000000000), + OllamaRunnerMemoryObservation { + realization: r_fixture_low, node: fixture_node, instrument: fixture_instrument, + context_depth: token_count(count: 400000), concurrent_sessions: 4, outcome: out_of_memory, + }, + ]) + is_admissible(verdict: evaluate_candidate( + candidate: mixed, node: fixture_node, constraints: floor_400k)) + && is_memory_rejection(verdict: evaluate_candidate( + candidate: mixed, node: fixture_node, constraints: floor_400k_four_sessions)) +} + test fn w_contradictory_attempts_refuse_in_both_orders_and_at_both_magnitudes() -> Bool { let big_success = fixture_footprint(r: r_fixture_low, resident: 99000000000) let small_failure = fixture_attempt(r: r_fixture_low, depth: 400000, outcome: out_of_memory) @@ -902,6 +1012,34 @@ fn is_admissible(verdict: CandidateVerdict) -> Bool { // A PREFILL READING FROM A QUIETER HOST DOES NOT ANSWER A BUSIER DEMAND. The rate axis gained the // same concurrency and node binding the memory axis already had, and for the measured reason: this // fleet read a queue on a serialized host as a 300x model latency regression. +// THE DISCRIMINATING PROBE FOR THE CONCURRENCY BINDING, and it is the direction the sibling witness +// above cannot see. That one asks a 1-session reading at 4 sessions, which fails under BOTH the old +// `>=` rule and the exact one -- it never discriminated. This one asks a 4-SESSION reading at a +// 1-session demand: under `>=` it QUALIFIED and answered 90 tok/s, on the assumption that a busier +// reading conservatively bounds a quieter one. That assumption is about a scheduler and is +// unmeasured, and if `rate` turns out to be an aggregate rather than per-request it points the +// other way. So it must now come back unanswerable. Flip the binding back to `>=` in +// slowest_fresh_rate_at_floor and this witness goes red. +data busy_only_prefill: List = [ + FreshPrefillObservation { realization: r_iq2_xxs, node: serving_node, concurrent_sessions: 4, depth: token_count(count: 400060), rate: tokens_per_second(count: 90) }, +] + +test fn w_a_busier_prefill_reading_does_not_answer_a_quieter_demand() -> Bool { + match prefill_rate_at_floor( + observations: busy_only_prefill, realization: r_iq2_xxs, + node: serving_node, hot_sessions: 1, + floor: token_count(count: 400000), regime: FreshSessionPrefill) { + Present { value: _ } => false + Absent => match prefill_rate_at_floor( + observations: busy_only_prefill, realization: r_iq2_xxs, + node: serving_node, hot_sessions: 4, + floor: token_count(count: 400000), regime: FreshSessionPrefill) { + Absent => false + Present { value: r } => tokens_per_second_count(r: r) == 90 + } + } +} + test fn w_a_prefill_reading_at_lower_concurrency_or_another_node_does_not_qualify() -> Bool { match quiet_rate_at(node: serving_node, hot: 1) { Absent => false @@ -944,23 +1082,46 @@ fn quiet_rate_at(node: NonEmptyStr, hot: Nat) -> TokensPerSecond? { // // The fixture below is exactly that pair: identical release, quantization, runtime and resolved // configuration; the artifact digest differs only in HASH FAMILY. -fn structural_digest(hex: String) -> ContentHash { - match content_hash_from_structural_digest(digest: hex) { - Present { value: h } => h - Absent => content_hash_of_value(value: "malformed structural digest" as NonEmptyStr) - } +// Total for the same reason `digest` above is: Fnv1a64StructuralDigestHex is a +// `String where lower_hex_16` refinement checked at typecheck, so there is no malformed arm to +// fabricate a value for. +fn structural_digest(hex: Fnv1a64StructuralDigestHex) -> ContentHash { + as_content_hash_structural(structural: structural_content_hash(digest: hex)) } fn r_iq2_xxs_structural_artifact() -> ServingRealizationIdentity { realization( release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, - artifact: structural_digest(hex: "32af248f4cab44ff"), + artifact: structural_digest(hex: "32af248f4cab44ff" as Fnv1a64StructuralDigestHex), runtime: ollama_runtime, - config: serial_config, + fixed_mode: serial_mode, ) } +// THE SCOPE OF THE STALL, which the witness above cannot see because its incomparable reading is +// also the RELEVANT one. Ruled in the side chat and it is a real defect, not a refinement: an +// incomparable axis may only stall a candidate when the observation carrying it would OTHERWISE HAVE +// PARTICIPATED. The predecessor filtered candidate evidence by node alone before asking whether +// identities compared, so a footprint taken at 8,192 tokens -- nowhere near a 400,000-token demand, +// and irrelevant on depth alone -- stalled the whole candidate because its digest happened to be +// recorded in a different hash family. Every extra reading on a node then made that node LESS +// answerable, which inverts what evidence is for. +// +// Here the same cross-family reading sits at a depth that cannot qualify, beside a good reading that +// can. The candidate must be ADMISSIBLE on the good reading; the irrelevant one stays irrelevant. +// Revert first_incomparable_axis to a node-only filter and this goes red. +test fn w_an_irrelevant_cross_family_reading_does_not_stall_the_candidate() -> Bool { + let mixed = with_memory_observations(base: build_iq2_xxs, observations: [ + served(r: r_iq2_xxs_structural_artifact(), node: serving_node, instrument: ps_instrument, + depth: 8192, sessions: 1, resident: 80000000000), + served(r: r_iq2_xxs, node: serving_node, instrument: ps_instrument, + depth: 400000, sessions: 1, resident: 87692389907), + ]) + is_admissible(verdict: evaluate_candidate( + candidate: mixed, node: serving_node, constraints: floor_400k)) +} + test fn w_a_cross_family_digest_refuses_rather_than_reading_as_unobserved() -> Bool { let cross_family = with_memory_observations(base: build_iq2_xxs, observations: [ served(r: r_iq2_xxs_structural_artifact(), node: serving_node, instrument: ps_instrument, @@ -990,9 +1151,9 @@ test fn w_realization_comparison_reaches_all_three_states_and_different_wins() - let other_release_cross_family = realization( release_id: qwen_release, quant: r_iq2_xxs.quant_label, - artifact: structural_digest(hex: "32af248f4cab44ff"), + artifact: structural_digest(hex: "32af248f4cab44ff" as Fnv1a64StructuralDigestHex), runtime: ollama_runtime, - config: serial_config, + fixed_mode: serial_mode, ) is_same(c: compare_serving_realization(a: r_iq2_xxs, b: r_iq2_xxs)) && is_different(c: compare_serving_realization(a: r_iq2_xxs, b: r_qwen)) @@ -1041,11 +1202,15 @@ test fn w_all_serving_choice_claims_hold() -> Bool { && w_a_prefill_receipt_from_another_realization_does_not_qualify() && w_only_a_typed_memory_refusal_rejects_on_memory() && w_contradictory_attempts_refuse_in_both_orders_and_at_both_magnitudes() + && w_a_success_at_the_floor_survives_an_oom_above_it() + && w_one_candidate_answers_admissible_quiet_and_rejected_busy() && w_the_large_buffer_success_alone_still_admits() && w_a_refusal_generalizes_to_no_easier_point_but_a_success_does() && w_the_concurrency_axis_carries_the_same_asymmetry() && w_a_prefill_reading_at_lower_concurrency_or_another_node_does_not_qualify() + && w_a_busier_prefill_reading_does_not_answer_a_quieter_demand() && w_a_cross_family_digest_refuses_rather_than_reading_as_unobserved() + && w_an_irrelevant_cross_family_reading_does_not_stall_the_candidate() && w_realization_comparison_reaches_all_three_states_and_different_wins() } From 931cd4c94167a4215fcc08b7279f044ef4f04695 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Wed, 2 Sep 2026 02:22:37 +0000 Subject: [PATCH 18/22] Held-but-unalignable is not absent, on the other two axes either The memory axis learned to distinguish "we hold evidence whose subject cannot be lined up" from "no evidence was taken". Semantic context and prefill had not: both still read serving_realization_equal, which folds RealizationIncomparable to false. So a retrieval receipt or a prefill reading whose artifact digest was recorded in a different hash family left the population silently, and the candidate reported SemanticContextUnverifiedAtFloor or PrefillRateUnmeasuredAtFloor -- telling the operator to measure something that has already been measured. Both now carry three states and surface the axis. MATERIALITY, NOT JUST RELEVANCE. Point relevance stopped irrelevant rows from stalling a candidate, but every point-relevant incomparable row still stalled it before anything asked whether resolving it could change the answer. A comparable 400k success beside a cross-family 1M success is decided either way -- second success or irrelevant row, fit holds -- and the code returned MemoryFitIdentityIncomparable anyway, so a candidate got LESS answerable for holding an extra success. An incomparable row now stalls exactly when its arm is the one that would decide. The converse still holds and is witnessed: an incomparable EXACT REFUSAL beside a success is load-bearing, because resolving its identity is what separates contradiction from fit. ONE ARGV, BOTH GRAINS. ExactRuntimeConfigurationIdentity was described in the source and absent from the types, so an author could derive a fixed mode from one argv and write any depth and sessions beside it. The observation carriers are now sole_constructor and mint the realization's fixed mode and the observation's exact configuration from a single argv list in one call. Not closed: the point values are not PARSED back out of the argv, because std has no string-to-integer admission and minting one here would be a second numeric authority. Stated in the source with that capability named as the next-rung trigger. ORDERING. The previous key tied on two failures differing only in cause, so shallowest_of kept whichever the roster listed first and the reported diagnosis reversed with the roster. Every verdict-visible payload is now in the key -- both byte axes numerically, the prose axis lexicographically, arm-tagged. Five new witnesses, each verified discriminating by reverting the rule it tests. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01U397y4s3dSBof7vGPAX87G --- dag/gunbc/model/choice.dag | 382 ++++++++++++++-- .../model/serving_choice_witness_test.dag | 413 ++++++++++++------ 2 files changed, 616 insertions(+), 179 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index cfae9c7d974..d826ee0b7dc 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -175,6 +175,41 @@ fn argv_digest_of(argv: List) -> Fnv1a64Structural { content_hash_combine_structural(left: acc, right: content_hash_atom(value: word))) } +// THE SECOND HALF OF THE ARGV SPLIT. FixedRuntimeModeIdentity answers "is this the same runner +// mode", which is what lets a reading taken at one point speak for another. It deliberately drops +// the varied flags, so by construction it CANNOT say which point a reading was taken at. +// +// That leaves a gap the source described and the types did not close: an author could derive a fixed +// mode from one argv and then write any context_depth and concurrent_sessions beside it. The tuple +// (fixed mode, authored depth, authored sessions) does not establish that all three came from one +// observed invocation. So the exact full-argv identity is carried BY THE OBSERVATION, and both it +// and the realization's fixed mode are minted from ONE argv list in one call -- they cannot be +// authored independently. +// +// WHAT THIS STILL DOES NOT DO, named rather than implied: the depth and session values are not +// PARSED out of the argv and checked against the authored ones. std has no string-to-integer +// admission, so there is nothing to parse them with, and inventing a parser inside this module would +// be a second numeric authority. The class therefore sits at mechanically-preventable rather than +// structurally-impossible, and its next-rung trigger is a std integer-admission capability +// sufficient to read a flag value back as a Nat -- at which point the point fields are derived from +// the argv and the mismatched state has no constructor. +type ExactRuntimeConfigurationIdentity sole_constructor { + argv_digest: Fnv1a64Structural +} + +fn exact_runtime_configuration_from_argv( + argv: List, +) -> ExactRuntimeConfigurationIdentity { + ExactRuntimeConfigurationIdentity { argv_digest: argv_digest_of(argv: argv) } +} + +fn exact_runtime_configuration_equal( + a: ExactRuntimeConfigurationIdentity, + b: ExactRuntimeConfigurationIdentity, +) -> Bool { + content_hash_eq_structural(left: a.argv_digest, right: b.argv_digest) +} + // The varied flags AND their values are DROPPED, not blanked, so the mode digest is invariant to // their values by construction rather than by the caller remembering to pass matching ones. // @@ -416,14 +451,40 @@ fn serving_regime_wire(regime: ServingRegime) -> String { // caller meets when three other sessions are prefilling: this fleet's own 300x latency scare was // exactly that -- a queue on a serialized host, read as a property of the model. And a rate measured // on one node does not describe another. Both now join. -type FreshPrefillObservation { +type FreshPrefillObservation sole_constructor { realization: ServingRealizationIdentity + configuration: ExactRuntimeConfigurationIdentity node: HostIdentity concurrent_sessions: Nat depth: TokenCount rate: TokensPerSecond } +fn observed_fresh_prefill( + release_id: ReleaseIdentity, + quant_label: NonEmptyStr, + artifact_digest: ContentHash, + runtime: ServingRuntimeIdentity, + argv: List, + varied_flags: List, + node: HostIdentity, + concurrent_sessions: Nat, + depth: TokenCount, + rate: TokensPerSecond, +) -> FreshPrefillObservation { + FreshPrefillObservation { + realization: serving_realization_identity( + release_id: release_id, quant_label: quant_label, artifact_digest: artifact_digest, + runtime: runtime, + fixed_mode: fixed_runtime_mode_from_argv(argv: argv, varied_flags: varied_flags)), + configuration: exact_runtime_configuration_from_argv(argv: argv), + node: node, + concurrent_sessions: concurrent_sessions, + depth: depth, + rate: rate, + } +} + // The slowest rate observed AT THE REQUESTED REGIME at or beyond the floor, or Absent when no such // observation exists. Slowest rather than fastest because a floor is a promise about the worst case // a caller will meet. @@ -434,24 +495,103 @@ type FreshPrefillObservation { // substituting a warm number for a fresh question is the dangerous direction and would admit one // that cannot. Neither substitution is available here: the non-fresh arms have no observation to // read at all, so the widening arm does not exist to be taken. -fn prefill_rate_at_floor( +// THE THIRD STATE ON THIS AXIS TOO. slowest_fresh_rate_at_floor filtered with +// serving_realization_equal, so a prefill receipt taken at exactly the right point on exactly the +// right host, whose artifact digest was recorded in a different hash family, silently left the +// population and the candidate reported PrefillRateUnmeasuredAtFloor. Held-but-unalignable is not +// unmeasured, and the distinction is the whole reason this module exists. +// +// The same materiality rule the memory axis uses applies here: an incomparable receipt stalls only +// when no COMPARABLE receipt already answers at this point. If a comparable reading exists, the +// answer is the slowest of those, and resolving the unalignable one could not change which floor +// question was answered. +type PrefillQualification + = PrefillRateQualified { rate: TokensPerSecond } + | PrefillIncomparable { axis: NonEmptyStr } + | PrefillUnmeasured + +fn fresh_prefill_at_point( + observations: List, + node: HostIdentity, + hot_sessions: Nat, + floor: TokenCount, +) -> List { + filter(observations, o => + host_identity_eq(a: o.node, b: node) + && o.concurrent_sessions == hot_sessions + && token_count_value(t: o.depth) >= token_count_value(t: floor)) +} + +fn first_prefill_incomparable_axis( + observations: List, + realization: ServingRealizationIdentity, +) -> NonEmptyStr? { + fold(observations, none, (found, o) => + match found { + Present { value: a } => Present { value: a } + Absent => incomparable_axis( + comparison: compare_serving_realization(a: o.realization, b: realization)) + }) +} + +fn fresh_prefill_qualification( + observations: List, + realization: ServingRealizationIdentity, + node: HostIdentity, + hot_sessions: Nat, + floor: TokenCount, +) -> PrefillQualification { + match slowest_fresh_rate_at_floor( + observations: observations, realization: realization, + node: node, hot_sessions: hot_sessions, floor: floor) { + Present { value: r } => PrefillRateQualified { rate: r } + Absent => + match first_prefill_incomparable_axis( + observations: fresh_prefill_at_point( + observations: observations, node: node, + hot_sessions: hot_sessions, floor: floor), + realization: realization) { + Present { value: a } => PrefillIncomparable { axis: a } + Absent => PrefillUnmeasured + } + } +} + +fn prefill_qualification_at_floor( observations: List, realization: ServingRealizationIdentity, node: HostIdentity, hot_sessions: Nat, floor: TokenCount, regime: ServingRegime, -) -> TokensPerSecond? { +) -> PrefillQualification { match regime { - WarmContinuation => none - RestoredSession => none + WarmContinuation => PrefillUnmeasured + RestoredSession => PrefillUnmeasured FreshSessionPrefill => - slowest_fresh_rate_at_floor( + fresh_prefill_qualification( observations: observations, realization: realization, node: node, hot_sessions: hot_sessions, floor: floor) } } +fn prefill_rate_at_floor( + observations: List, + realization: ServingRealizationIdentity, + node: HostIdentity, + hot_sessions: Nat, + floor: TokenCount, + regime: ServingRegime, +) -> TokensPerSecond? { + match prefill_qualification_at_floor( + observations: observations, realization: realization, node: node, + hot_sessions: hot_sessions, floor: floor, regime: regime) { + PrefillRateQualified { rate: r } => Present { value: r } + PrefillIncomparable { axis: _ } => none + PrefillUnmeasured => none + } +} + // DEPTH CARRIES A DIRECTION. CONCURRENCY DOES NOT -- NOT YET, AND THE DIFFERENCE IS EVIDENCE. // // Depth qualifies conservatively: a rate measured DEEPER bounds the rate at a shallower demand, @@ -552,8 +692,9 @@ type RunnerAttemptOutcome | RunnerRefusedForMemory { detail: NonEmptyStr } | RunnerFailedForOtherCause { cause: NonEmptyStr } -type OllamaRunnerMemoryObservation { +type OllamaRunnerMemoryObservation sole_constructor { realization: ServingRealizationIdentity + configuration: ExactRuntimeConfigurationIdentity node: HostIdentity instrument: NonEmptyStr context_depth: TokenCount @@ -561,6 +702,37 @@ type OllamaRunnerMemoryObservation { outcome: RunnerAttemptOutcome } +// ONE ARGV, BOTH GRAINS. The caller hands over the argv the runner was actually launched with and +// the flags it varies; the fixed mode inside the realization and the exact configuration on the +// observation are both derived here, from that one list. There is no surface on which they can +// disagree, because there is no surface on which they are separately supplied. +fn observed_memory_attempt( + release_id: ReleaseIdentity, + quant_label: NonEmptyStr, + artifact_digest: ContentHash, + runtime: ServingRuntimeIdentity, + argv: List, + varied_flags: List, + node: HostIdentity, + instrument: NonEmptyStr, + context_depth: TokenCount, + concurrent_sessions: Nat, + outcome: RunnerAttemptOutcome, +) -> OllamaRunnerMemoryObservation { + OllamaRunnerMemoryObservation { + realization: serving_realization_identity( + release_id: release_id, quant_label: quant_label, artifact_digest: artifact_digest, + runtime: runtime, + fixed_mode: fixed_runtime_mode_from_argv(argv: argv, varied_flags: varied_flags)), + configuration: exact_runtime_configuration_from_argv(argv: argv), + node: node, + instrument: instrument, + context_depth: context_depth, + concurrent_sessions: concurrent_sessions, + outcome: outcome, + } +} + // WHAT THE QUALIFYING ATTEMPTS SAY, reconciled by TYPE rather than by byte magnitude. // // The previous version folded every qualifying observation to the smallest reported buffer and only @@ -712,7 +884,49 @@ fn participates_at_point( served: generalizes, refused_for_memory: exact, failed_otherwise: exact) } -fn first_incomparable_axis( +// POINT RELEVANCE WAS NECESSARY BUT NOT SUFFICIENT: the row must also be able to CHANGE THE ANSWER. +// +// The first repair here stopped irrelevant rows from stalling a candidate. It still stalled on every +// point-relevant incomparable row, before asking whether resolving that row could alter anything. +// The counterexample is ordinary: a comparable 400k one-session SUCCESS beside a cross-family 1M +// one-session SUCCESS, asked at 400k. If the second row describes this candidate it is merely a +// second success; if it describes another realization it is irrelevant. Fit is established either +// way, so the identity question has no bearing -- and the code returned MemoryFitIdentityIncomparable +// anyway. That is the same anti-monotone shape as before, one level in: a candidate whose fit is +// already established got LESS answerable for holding an extra success. +// +// So an incomparable row stalls exactly when its arm is the one that would decide, which is the arm +// with no comparable row already in it: +// +// incomparable SUCCESS -- stalls only when no comparable success exists, because it could turn +// Unobserved or RefusedForMemory into Established or Contradicted. +// incomparable REFUSAL -- stalls only when no comparable refusal exists, because it could turn +// Established into Contradicted, or Unobserved into RefusedForMemory. +// incomparable FAILURE -- stalls only when nothing comparable was observed at all, since a +// non-memory failure is read only when the other two arms are empty. +// +// This is not "ignore incomparable evidence". An incomparable exact refusal beside a comparable +// success is still load-bearing and still stalls, because resolving its identity is exactly what +// separates a contradiction from ordinary fit. +fn incomparable_axis_of( + o: OllamaRunnerMemoryObservation, + realization: ServingRealizationIdentity, +) -> NonEmptyStr? { + incomparable_axis(comparison: compare_serving_realization(a: o.realization, b: realization)) +} + +fn first_axis_in( + observations: List, + realization: ServingRealizationIdentity, +) -> NonEmptyStr? { + fold(observations, none, (found, o) => + match found { + Present { value: a } => Present { value: a } + Absent => incomparable_axis_of(o: o, realization: realization) + }) +} + +fn deciding_incomparable_axis( observations: List, realization: ServingRealizationIdentity, floor: TokenCount, @@ -722,12 +936,36 @@ fn first_incomparable_axis( let relevant = filter(observations, o => host_identity_eq(a: o.node, b: node) && participates_at_point(o: o, floor: floor, hot_sessions: hot_sessions)) - fold(relevant, none, (found, o) => - match found { - Present { value: a } => Present { value: a } - Absent => incomparable_axis( - comparison: compare_serving_realization(a: o.realization, b: realization)) - }) + let served_rows = filter(relevant, o => outcome_is_served(o: o)) + let refused_rows = filter(relevant, o => outcome_is_memory_refusal(o: o)) + let broken_rows = filter(relevant, o => outcome_is_other_failure(o: o)) + let comparable_served = served_qualifying_observations( + observations: observations, realization: realization, + floor: floor, hot_sessions: hot_sessions, node: node) + let comparable_refused = refused_at_exact_point( + observations: observations, realization: realization, + floor: floor, hot_sessions: hot_sessions, node: node) + let comparable_broken = failed_otherwise_at_exact_point( + observations: observations, realization: realization, + floor: floor, hot_sessions: hot_sessions, node: node) + let from_served = + if length(comparable_served) > 0 { none } + else { first_axis_in(observations: served_rows, realization: realization) } + let from_refused = + if length(comparable_refused) > 0 { none } + else { first_axis_in(observations: refused_rows, realization: realization) } + let from_broken = + if length(comparable_served) > 0 || length(comparable_refused) > 0 + || length(comparable_broken) > 0 { none } + else { first_axis_in(observations: broken_rows, realization: realization) } + match from_served { + Present { value: a } => Present { value: a } + Absent => + match from_refused { + Present { value: a } => Present { value: a } + Absent => from_broken + } + } } // The SHALLOWEST qualifying attempt, because it is the one closest to the configuration actually @@ -755,18 +993,46 @@ fn observation_order_key_before( if a.concurrent_sessions != b.concurrent_sessions { a.concurrent_sessions < b.concurrent_sessions } else { - let ab = reported_total_bytes(o: a) - let bb = reported_total_bytes(o: b) - if ab != bb { ab < bb } else { - (a.instrument as String) < (b.instrument as String) + let ai = a.instrument as String + let bi = b.instrument as String + if ai != bi { ai < bi } else { + let at = reported_total_bytes(o: a) + let bt = reported_total_bytes(o: b) + if at != bt { at < bt } else { + let ag = reported_gpu_bytes(o: a) + let bg = reported_gpu_bytes(o: b) + if ag != bg { ag < bg } else { + outcome_detail_key(o: a) < outcome_detail_key(o: b) + } + } } } } } -// The footprint axis is only populated for a SERVED attempt; a refusal reports no buffer. Both -// non-served arms sort as zero, which is not a fabricated measurement -- the three arms never share -// a comparison, because memory_fit_evidence partitions by outcome before it orders. +// EVERY VERDICT-VISIBLE PAYLOAD IS IN THE KEY, because anything visible in the answer and absent +// from the key is a field the roster order gets to choose. +// +// The first repair here ordered on depth, sessions, reported TOTAL buffer and instrument, which +// closed the depth-only tie but left a narrower one open. Two non-memory failures at the same exact +// point, same realization, host and instrument, differing only in CAUSE, produce an identical key; +// shallowest_of keeps whichever the roster listed first, and attempt_failure_cause then reports that +// cause in the MemoryAttemptFailedForNonMemoryCause diagnosis. Reverse the roster, reverse the +// diagnosis. The same hole existed for two served readings agreeing on total buffer and differing on +// the GPU buffer, and for two refusals differing only in detail. +// +// So the outcome contributes its WHOLE payload as the final key, arm-tagged so the three arms cannot +// collide across a shared spelling. This is not a fabricated ordering over incommensurable things: +// the arms are already partitioned by memory_fit_evidence before any comparison happens, so the tag +// only ever breaks ties WITHIN an arm. Two observations equal on this key are equal on every field +// the verdict can show, so which one is returned genuinely carries no information -- which is the +// property the previous key claimed and did not have. +// The two byte axes compare NUMERICALLY and the prose axis lexicographically, rather than folding +// everything into one string: string order over decimal digits puts "9" after "10", which would be a +// deterministic but wrong magnitude ordering, and the whole point of ordering on footprint is that +// the smaller reading is the closer one. Non-served arms report no buffer and contribute zero, which +// is not a fabricated measurement -- memory_fit_evidence partitions by outcome before any comparison +// happens, so a served and a refused reading are never compared against each other. fn reported_total_bytes(o: OllamaRunnerMemoryObservation) -> Int { match o.outcome { RunnerServed { reported_total_buffer: t, reported_gpu_buffer: _ } => byte_size_count(b: t) @@ -775,6 +1041,24 @@ fn reported_total_bytes(o: OllamaRunnerMemoryObservation) -> Int { } } +fn reported_gpu_bytes(o: OllamaRunnerMemoryObservation) -> Int { + match o.outcome { + RunnerServed { reported_total_buffer: _, reported_gpu_buffer: g } => byte_size_count(b: g) + RunnerRefusedForMemory { detail: _ } => 0 + RunnerFailedForOtherCause { cause: _ } => 0 + } +} + +// Arm-tagged so a refusal detail and a failure cause that happen to share a spelling still order +// apart, even though the partition means they are never compared in practice. +fn outcome_detail_key(o: OllamaRunnerMemoryObservation) -> String { + match o.outcome { + RunnerServed { reported_total_buffer: _, reported_gpu_buffer: _ } => "served" + RunnerRefusedForMemory { detail: d } => join(["refused|", d as String], "") + RunnerFailedForOtherCause { cause: c } => join(["failed|", c as String], "") + } +} + fn shallowest( a: OllamaRunnerMemoryObservation, b: OllamaRunnerMemoryObservation, @@ -800,7 +1084,7 @@ fn memory_fit_evidence( hot_sessions: Nat, node: HostIdentity, ) -> MemoryFitEvidence { - match first_incomparable_axis( + match deciding_incomparable_axis( observations: observations, realization: realization, floor: floor, hot_sessions: hot_sessions, node: node) { Present { value: axis } => MemoryFitIdentityIncomparable { axis: axis } @@ -1009,21 +1293,24 @@ fn evaluate_candidate( MemoryFitRefusedForMemory { by: r } => rejected(candidate: candidate, axis: DoesNotFitMemory { observed: r }) MemoryFitEstablished { by: observed } => - match semantic_context_verified_to( + match semantic_context_qualification( evidence: candidate.semantic_context, realization: candidate.realization, ) { - Absent => unanswerable(candidate: candidate, missing: SemanticContextUnverifiedAtFloor { - floor: constraints.context_floor, - declared: candidate.declared_context, - }) - Present { value: verified } => + SemanticContextIncomparable { axis: a } => + unanswerable(candidate: candidate, missing: EvidenceIdentityIncomparable { axis: a }) + SemanticContextNotForThisRealization => + unanswerable(candidate: candidate, missing: SemanticContextUnverifiedAtFloor { + floor: constraints.context_floor, + declared: candidate.declared_context, + }) + SemanticContextQualified { verified_to: verified } => match token_count_value(t: verified) < token_count_value(t: constraints.context_floor) { true => rejected(candidate: candidate, axis: ContextBelowFloor { declared: verified, floor: constraints.context_floor, }) - false => match prefill_rate_at_floor( + false => match prefill_qualification_at_floor( observations: candidate.fresh_prefill_observations, realization: candidate.realization, node: node, @@ -1031,11 +1318,14 @@ fn evaluate_candidate( floor: constraints.context_floor, regime: constraints.prefill_regime, ) { - Absent => unanswerable(candidate: candidate, missing: PrefillRateUnmeasuredAtFloor { - floor: constraints.context_floor, - regime: constraints.prefill_regime, - }) - Present { value: prefill } => + PrefillIncomparable { axis: a } => + unanswerable(candidate: candidate, missing: EvidenceIdentityIncomparable { axis: a }) + PrefillUnmeasured => + unanswerable(candidate: candidate, missing: PrefillRateUnmeasuredAtFloor { + floor: constraints.context_floor, + regime: constraints.prefill_regime, + }) + PrefillRateQualified { rate: prefill } => match tokens_per_second_count(r: prefill) < tokens_per_second_count(r: constraints.prefill_floor) { true => rejected(candidate: candidate, axis: PrefillBelowFloor { measured: prefill, @@ -1060,16 +1350,28 @@ fn attempt_failure_cause(observation: OllamaRunnerMemoryObservation) -> NonEmpty // The retrieval receipt counts only when it was taken against THIS realization. A depth verified on // a different quantization or a different artifact is a fact about that realization, and reading it // here would let a candidate borrow evidence it never earned. -fn semantic_context_verified_to( +// THREE STATES, BECAUSE THE BOOLEAN PROJECTION LOST THE ONE THAT MATTERS. This read +// serving_realization_equal, which folds RealizationIncomparable to `false` -- so a retrieval +// receipt whose artifact digest was recorded in a different hash family came back Absent and the +// candidate reported SemanticContextUnverifiedAtFloor. That is the defect the memory axis already +// closed, surviving on a second axis: "we hold a receipt we cannot line up" rendered as "no receipt +// was taken". The operator is told to go measure something that has already been measured. +type SemanticContextQualification + = SemanticContextQualified { verified_to: TokenCount } + | SemanticContextIncomparable { axis: NonEmptyStr } + | SemanticContextNotForThisRealization + +fn semantic_context_qualification( evidence: SemanticContextEvidence?, realization: ServingRealizationIdentity, -) -> TokenCount? { +) -> SemanticContextQualification { match evidence { - Absent => none + Absent => SemanticContextNotForThisRealization Present { value: e } => - match serving_realization_equal(a: e.realization, b: realization) { - true => Present { value: e.verified_to } - false => none + match compare_serving_realization(a: e.realization, b: realization) { + RealizationSame => SemanticContextQualified { verified_to: e.verified_to } + RealizationDifferent => SemanticContextNotForThisRealization + RealizationIncomparable { axis: a } => SemanticContextIncomparable { axis: a } } } } diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index 1a91dfdb004..94dac4843a7 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -131,28 +131,22 @@ fn realization( } fn served( - r: ServingRealizationIdentity, - node: NonEmptyStr, + b: FixtureBuild, + node: HostIdentity, instrument: NonEmptyStr, depth: Nat, sessions: Nat, resident: Nat, ) -> OllamaRunnerMemoryObservation { - OllamaRunnerMemoryObservation { - realization: r, - node: node, - instrument: instrument, - context_depth: token_count(count: depth), - concurrent_sessions: sessions, + attempt_of(b: b, node: node, instrument: instrument, depth: depth, sessions: sessions, outcome: RunnerServed { reported_total_buffer: byte_size(count: resident), reported_gpu_buffer: byte_size(count: resident), - }, - } + }) } fn iq2_xxs_footprint(depth: Nat, resident: Nat) -> OllamaRunnerMemoryObservation { - served(r: r_iq2_xxs, node: serving_node, instrument: ps_instrument, + served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: depth, sessions: 1, resident: resident) } @@ -178,54 +172,91 @@ data qwen_release: ReleaseIdentity = ReleaseIdentity { revision: "35B" as NonEmptyStr, } -data r_iq2_xxs: ServingRealizationIdentity = realization( - release_id: deepseek_v4_flash_release, - quant: "IQ2_XXS" as NonEmptyStr, - artifact: digest(hex: "32af248f4cab44ffcb631298f66ff0bdf92a9350a3763a86b9a5b304fc4c1e64" as Sha256DigestHex), - runtime: ollama_runtime, - fixed_mode: serial_mode, -) +// A BUILD, NOT A REALIZATION. The observation carriers are now sole_constructor and mint the +// realization's fixed mode and the observation's exact configuration from ONE argv, so a fixture +// cannot hand them different ones. That means the fixture's unit is the BUILD -- components plus the +// argv the runner was launched with -- and both the realization and every reading derive from it. +type FixtureBuild { + release_id: ReleaseIdentity + quant: NonEmptyStr + artifact: ContentHash + runtime: ServingRuntimeIdentity + argv: List +} -data r_iq3_s: ServingRealizationIdentity = realization( - release_id: deepseek_v4_flash_release, - quant: "UD-IQ3_S" as NonEmptyStr, - artifact: digest(hex: "5b7e1042cc9d3fa6081b4e2d7c3950ae61f8d24b0937ce5182a4f6b0d3e79c18" as Sha256DigestHex), - runtime: ollama_runtime, - fixed_mode: serial_mode, -) +fn realization_of(b: FixtureBuild) -> ServingRealizationIdentity { + serving_realization_identity( + release_id: b.release_id, quant_label: b.quant, artifact_digest: b.artifact, + runtime: b.runtime, + fixed_mode: fixed_runtime_mode_from_argv(argv: b.argv, varied_flags: varied_flags)) +} -data r_unmeasured: ServingRealizationIdentity = realization( - release_id: deepseek_v4_flash_release, - quant: "UD-IQ4_XS" as NonEmptyStr, - artifact: digest(hex: "9d2c48b6103fae57c81d0b2e64937fa5081c3d6e9b204f7a15c8e0b3d629f47a" as Sha256DigestHex), - runtime: ollama_runtime, - fixed_mode: serial_mode, -) +fn attempt_of( + b: FixtureBuild, + node: HostIdentity, + instrument: NonEmptyStr, + depth: Nat, + sessions: Nat, + outcome: RunnerAttemptOutcome, +) -> OllamaRunnerMemoryObservation { + observed_memory_attempt( + release_id: b.release_id, quant_label: b.quant, artifact_digest: b.artifact, + runtime: b.runtime, argv: b.argv, varied_flags: varied_flags, + node: node, instrument: instrument, + context_depth: token_count(count: depth), concurrent_sessions: sessions, outcome: outcome) +} -data r_qwen: ServingRealizationIdentity = realization( - release_id: qwen_release, - quant: "Q8_0" as NonEmptyStr, - artifact: digest(hex: "a41f70c39b2d85e6014c7f3a92b60d58e7139fac402b6d81c53e97a0f42b18d6" as Sha256DigestHex), - runtime: ollama_runtime, - fixed_mode: serial_mode, -) +fn prefill_of( + b: FixtureBuild, + node: HostIdentity, + sessions: Nat, + depth: Nat, + rate: Nat, +) -> FreshPrefillObservation { + observed_fresh_prefill( + release_id: b.release_id, quant_label: b.quant, artifact_digest: b.artifact, + runtime: b.runtime, argv: b.argv, varied_flags: varied_flags, + node: node, concurrent_sessions: sessions, + depth: token_count(count: depth), rate: tokens_per_second(count: rate)) +} + +data serving_argv: List = [ + "ollama" as NonEmptyStr, "serve" as NonEmptyStr, + "--num-ctx" as NonEmptyStr, "400000" as NonEmptyStr, + "--parallel" as NonEmptyStr, "1" as NonEmptyStr, +] -fn fixture_realization(quant: NonEmptyStr) -> ServingRealizationIdentity { - realization( - release_id: fixture_release, - quant: quant, +data fixture_argv: List = [ + "fixture-runner" as NonEmptyStr, "--fixture" as NonEmptyStr, +] + +data b_iq2_xxs: FixtureBuild = FixtureBuild { release_id: deepseek_v4_flash_release, quant: "IQ2_XXS" as NonEmptyStr, artifact: digest(hex: "32af248f4cab44ffcb631298f66ff0bdf92a9350a3763a86b9a5b304fc4c1e64" as Sha256DigestHex), runtime: ollama_runtime, argv: serving_argv } +data b_iq3_s: FixtureBuild = FixtureBuild { release_id: deepseek_v4_flash_release, quant: "UD-IQ3_S" as NonEmptyStr, artifact: digest(hex: "5b7e1042cc9d3fa6081b4e2d7c3950ae61f8d24b0937ce5182a4f6b0d3e79c18" as Sha256DigestHex), runtime: ollama_runtime, argv: serving_argv } +data b_unmeasured: FixtureBuild = FixtureBuild { release_id: deepseek_v4_flash_release, quant: "UD-IQ4_XS" as NonEmptyStr, artifact: digest(hex: "9d2c48b6103fae57c81d0b2e64937fa5081c3d6e9b204f7a15c8e0b3d629f47a" as Sha256DigestHex), runtime: ollama_runtime, argv: serving_argv } +data b_qwen: FixtureBuild = FixtureBuild { release_id: qwen_release, quant: "Q8_0" as NonEmptyStr, artifact: digest(hex: "a41f70c39b2d85e6014c7f3a92b60d58e7139fac402b6d81c53e97a0f42b18d6" as Sha256DigestHex), runtime: ollama_runtime, argv: serving_argv } + +fn fixture_build(quant: NonEmptyStr) -> FixtureBuild { + FixtureBuild { + release_id: fixture_release, quant: quant, artifact: digest(hex: "ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00" as Sha256DigestHex), runtime: serving_runtime_identity_for_release(rel: ollama_v0_32_9_jetpack6_release), - fixed_mode: fixed_runtime_mode_from_argv( - argv: ["fixture-runner" as NonEmptyStr, "--fixture" as NonEmptyStr], - varied_flags: varied_flags), - ) + argv: fixture_argv, + } } -data r_fixture_low: ServingRealizationIdentity = fixture_realization(quant: "LOW" as NonEmptyStr) -data r_fixture_high: ServingRealizationIdentity = fixture_realization(quant: "HIGH" as NonEmptyStr) -data r_tied_a: ServingRealizationIdentity = fixture_realization(quant: "TIED-A" as NonEmptyStr) -data r_tied_b: ServingRealizationIdentity = fixture_realization(quant: "TIED-B" as NonEmptyStr) +data b_fixture_low: FixtureBuild = fixture_build(quant: "LOW" as NonEmptyStr) +data b_fixture_high: FixtureBuild = fixture_build(quant: "HIGH" as NonEmptyStr) +data b_tied_a: FixtureBuild = fixture_build(quant: "TIED-A" as NonEmptyStr) +data b_tied_b: FixtureBuild = fixture_build(quant: "TIED-B" as NonEmptyStr) + +data r_iq2_xxs: ServingRealizationIdentity = realization_of(b: b_iq2_xxs) +data r_iq3_s: ServingRealizationIdentity = realization_of(b: b_iq3_s) +data r_unmeasured: ServingRealizationIdentity = realization_of(b: b_unmeasured) +data r_qwen: ServingRealizationIdentity = realization_of(b: b_qwen) +data r_fixture_low: ServingRealizationIdentity = realization_of(b: b_fixture_low) +data r_fixture_high: ServingRealizationIdentity = realization_of(b: b_fixture_high) +data r_tied_a: ServingRealizationIdentity = realization_of(b: b_tied_a) +data r_tied_b: ServingRealizationIdentity = realization_of(b: b_tied_b) fn verified_at(r: ServingRealizationIdentity, depth: Nat) -> SemanticContextEvidence? { Present { value: SemanticContextEvidence { @@ -243,9 +274,9 @@ data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_iq2_xxs, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_iq2_xxs, node: serving_node, concurrent_sessions: 1, depth: token_count(count: 160060), rate: tokens_per_second(count: 253) }, - FreshPrefillObservation { realization: r_iq2_xxs, node: serving_node, concurrent_sessions: 1, depth: token_count(count: 255061), rate: tokens_per_second(count: 197) }, - FreshPrefillObservation { realization: r_iq2_xxs, node: serving_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, + prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 160060, rate: 253), + prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 255061, rate: 197), + prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135), ], runtime_memory_observations: [ iq2_xxs_footprint(depth: 131072, resident: 86532465622), @@ -262,7 +293,7 @@ data build_iq3_s: QuantizedCandidate = QuantizedCandidate { semantic_context: none, fresh_prefill_observations: [], runtime_memory_observations: [ - served(r: r_iq3_s, node: serving_node, instrument: ps_instrument, + served(b: b_iq3_s, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 116970000000), ], } @@ -524,8 +555,8 @@ test fn w_a_prefill_receipt_from_another_realization_does_not_qualify() -> Bool // one, and the ordering claims they support are about the SELECTOR and not about any hardware. data fixture_instrument: NonEmptyStr = "declared fixture, not a reading" as NonEmptyStr -fn fixture_footprint(r: ServingRealizationIdentity, resident: Nat) -> OllamaRunnerMemoryObservation { - served(r: r, node: fixture_node, instrument: fixture_instrument, +fn fixture_footprint(b: FixtureBuild, resident: Nat) -> OllamaRunnerMemoryObservation { + served(b: b, node: fixture_node, instrument: fixture_instrument, depth: 1048576, sessions: 1, resident: resident) } @@ -533,18 +564,12 @@ fn fixture_footprint(r: ServingRealizationIdentity, resident: Nat) -> OllamaRunn // refuses at 1,048,576 cannot be pointed at a 400,000 demand and expected to reject it -- that is // the very inference the module now declines to draw. fn fixture_attempt( - r: ServingRealizationIdentity, + b: FixtureBuild, depth: Nat, outcome: RunnerAttemptOutcome, ) -> OllamaRunnerMemoryObservation { - OllamaRunnerMemoryObservation { - realization: r, - node: fixture_node, - instrument: fixture_instrument, - context_depth: token_count(count: depth), - concurrent_sessions: 1, - outcome: outcome, - } + attempt_of(b: b, node: fixture_node, instrument: fixture_instrument, + depth: depth, sessions: 1, outcome: outcome) } // These two are declared fixtures exercising the ordering rule, NOT fleet observations. They share a @@ -556,9 +581,9 @@ data fixture_low_rank: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_fixture_low, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_fixture_low, node: fixture_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + prefill_of(b: b_fixture_low, node: fixture_node, sessions: 1, depth: 400060, rate: 500), ], - runtime_memory_observations: [fixture_footprint(r: r_fixture_low, resident: 28000000000)], + runtime_memory_observations: [fixture_footprint(b: b_fixture_low, resident: 28000000000)], } data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { @@ -567,9 +592,9 @@ data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_fixture_high, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_fixture_high, node: fixture_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 400) }, + prefill_of(b: b_fixture_high, node: fixture_node, sessions: 1, depth: 400060, rate: 400), ], - runtime_memory_observations: [fixture_footprint(r: r_fixture_high, resident: 38000000000)], + runtime_memory_observations: [fixture_footprint(b: b_fixture_high, resident: 38000000000)], } test fn w_the_higher_quality_candidate_wins_when_both_are_admissible() -> Bool { @@ -606,13 +631,7 @@ test fn w_a_declared_context_never_qualifies_without_a_retrieval_receipt() -> Bo // (253 -> 197 -> 135 measured on one realization), so a shallow reading flatters the candidate. test fn w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() -> Bool { match prefill_rate_at_floor( - observations: [FreshPrefillObservation { - realization: r_iq2_xxs, - node: serving_node, - concurrent_sessions: 1, - depth: token_count(count: 160060), - rate: tokens_per_second(count: 253), - }], + observations: [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 160060, rate: 253)], realization: r_iq2_xxs, node: serving_node, hot_sessions: 1, @@ -694,13 +713,7 @@ test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { // regime field at all, so a warm rate is not merely unmatched here, it is unwritable anywhere. The // non-fresh arms of prefill_rate_at_floor have nothing to read. test fn w_a_fresh_prefill_receipt_does_not_answer_a_warm_continuation_floor() -> Bool { - let deep_fresh = [FreshPrefillObservation { - realization: r_iq2_xxs, - node: serving_node, - concurrent_sessions: 1, - depth: token_count(count: 400060), - rate: tokens_per_second(count: 135), - }] + let deep_fresh = [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135)] match prefill_rate_at_floor( observations: deep_fresh, realization: r_iq2_xxs, node: serving_node, hot_sessions: 1, @@ -775,12 +788,12 @@ data out_of_memory: RunnerAttemptOutcome = RunnerRefusedForMemory { fn oom_low() -> QuantizedCandidate { with_memory_observations(base: fixture_low_rank, - observations: [fixture_attempt(r: r_fixture_low, depth: 400000, outcome: out_of_memory)]) + observations: [fixture_attempt(b: b_fixture_low, depth: 400000, outcome: out_of_memory)]) } fn oom_high() -> QuantizedCandidate { with_memory_observations(base: fixture_high_rank, - observations: [fixture_attempt(r: r_fixture_high, depth: 400000, outcome: out_of_memory)]) + observations: [fixture_attempt(b: b_fixture_high, depth: 400000, outcome: out_of_memory)]) } // ================= WHAT AN UNSUCCESSFUL ATTEMPT MEANS ================= @@ -791,7 +804,7 @@ fn oom_high() -> QuantizedCandidate { // other failure is UNANSWERABLE and names its cause. test fn w_only_a_typed_memory_refusal_rejects_on_memory() -> Bool { let broken = with_memory_observations(base: fixture_low_rank, observations: [ - fixture_attempt(r: r_fixture_low, depth: 400000, outcome: RunnerFailedForOtherCause { + fixture_attempt(b: b_fixture_low, depth: 400000, outcome: RunnerFailedForOtherCause { cause: "chat template rendering failed" as NonEmptyStr, }), ]) @@ -854,12 +867,12 @@ fn is_contradiction_refusal(verdict: CandidateVerdict) -> Bool { // visible: a rule that qualified refusals by `depth >= floor` instead of `depth == floor` passes // every single-observation witness and fails only here. fn fixture_served_at( - r: ServingRealizationIdentity, + b: FixtureBuild, depth: Nat, sessions: Nat, resident: Nat, ) -> OllamaRunnerMemoryObservation { - served(r: r, node: fixture_node, instrument: fixture_instrument, + served(b: b, node: fixture_node, instrument: fixture_instrument, depth: depth, sessions: sessions, resident: resident) } @@ -871,8 +884,8 @@ fn fixture_served_at( test fn w_a_success_at_the_floor_survives_an_oom_above_it() -> Bool { is_admissible(verdict: evaluate_candidate( candidate: with_memory_observations(base: fixture_low_rank, observations: [ - fixture_served_at(r: r_fixture_low, depth: 400000, sessions: 1, resident: 28000000000), - fixture_attempt(r: r_fixture_low, depth: 1048576, outcome: out_of_memory), + fixture_served_at(b: b_fixture_low, depth: 400000, sessions: 1, resident: 28000000000), + fixture_attempt(b: b_fixture_low, depth: 1048576, outcome: out_of_memory), ]), node: fixture_node, constraints: floor_400k)) } @@ -885,11 +898,8 @@ test fn w_a_success_at_the_floor_survives_an_oom_above_it() -> Bool { // ignored the concurrency axis on either filter would have to give both demands the same verdict. test fn w_one_candidate_answers_admissible_quiet_and_rejected_busy() -> Bool { let mixed = with_memory_observations(base: fixture_low_rank, observations: [ - fixture_served_at(r: r_fixture_low, depth: 400000, sessions: 1, resident: 28000000000), - OllamaRunnerMemoryObservation { - realization: r_fixture_low, node: fixture_node, instrument: fixture_instrument, - context_depth: token_count(count: 400000), concurrent_sessions: 4, outcome: out_of_memory, - }, + fixture_served_at(b: b_fixture_low, depth: 400000, sessions: 1, resident: 28000000000), + attempt_of(b: b_fixture_low, node: fixture_node, instrument: fixture_instrument, depth: 400000, sessions: 4, outcome: out_of_memory), ]) is_admissible(verdict: evaluate_candidate( candidate: mixed, node: fixture_node, constraints: floor_400k)) @@ -898,8 +908,8 @@ test fn w_one_candidate_answers_admissible_quiet_and_rejected_busy() -> Bool { } test fn w_contradictory_attempts_refuse_in_both_orders_and_at_both_magnitudes() -> Bool { - let big_success = fixture_footprint(r: r_fixture_low, resident: 99000000000) - let small_failure = fixture_attempt(r: r_fixture_low, depth: 400000, outcome: out_of_memory) + let big_success = fixture_footprint(b: b_fixture_low, resident: 99000000000) + let small_failure = fixture_attempt(b: b_fixture_low, depth: 400000, outcome: out_of_memory) is_contradiction_refusal(verdict: evaluate_candidate( candidate: with_memory_observations(base: fixture_low_rank, observations: [big_success, small_failure]), @@ -915,7 +925,7 @@ test fn w_contradictory_attempts_refuse_in_both_orders_and_at_both_magnitudes() test fn w_the_large_buffer_success_alone_still_admits() -> Bool { match evaluate_candidate( candidate: with_memory_observations(base: fixture_low_rank, - observations: [fixture_footprint(r: r_fixture_low, resident: 99000000000)]), + observations: [fixture_footprint(b: b_fixture_low, resident: 99000000000)]), node: fixture_node, constraints: floor_400k) { ServingCandidateAdmissible { candidate: _, fit: _ } => true ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false @@ -938,11 +948,11 @@ test fn w_the_large_buffer_success_alone_still_admits() -> Bool { // more of the same machine. This is the direction that carries, and it must still carry. test fn w_a_refusal_generalizes_to_no_easier_point_but_a_success_does() -> Bool { let refused_deep = with_memory_observations(base: fixture_low_rank, - observations: [fixture_attempt(r: r_fixture_low, depth: 1048576, outcome: out_of_memory)]) + observations: [fixture_attempt(b: b_fixture_low, depth: 1048576, outcome: out_of_memory)]) let refused_at_point = with_memory_observations(base: fixture_low_rank, - observations: [fixture_attempt(r: r_fixture_low, depth: 400000, outcome: out_of_memory)]) + observations: [fixture_attempt(b: b_fixture_low, depth: 400000, outcome: out_of_memory)]) let served_deep = with_memory_observations(base: fixture_low_rank, - observations: [fixture_footprint(r: r_fixture_low, resident: 28000000000)]) + observations: [fixture_footprint(b: b_fixture_low, resident: 28000000000)]) is_unobserved_verdict(verdict: evaluate_candidate( candidate: refused_deep, node: fixture_node, constraints: floor_400k)) && is_memory_rejection(verdict: evaluate_candidate( @@ -955,20 +965,11 @@ test fn w_a_refusal_generalizes_to_no_easier_point_but_a_success_does() -> Bool // nothing about one; a success at four establishes one. test fn w_the_concurrency_axis_carries_the_same_asymmetry() -> Bool { let refused_at_four = with_memory_observations(base: fixture_low_rank, observations: [ - OllamaRunnerMemoryObservation { - realization: r_fixture_low, node: fixture_node, instrument: fixture_instrument, - context_depth: token_count(count: 400000), concurrent_sessions: 4, outcome: out_of_memory, - }, + attempt_of(b: b_fixture_low, node: fixture_node, instrument: fixture_instrument, depth: 400000, sessions: 4, outcome: out_of_memory), ]) let served_at_four = with_memory_observations(base: fixture_low_rank, observations: [ - OllamaRunnerMemoryObservation { - realization: r_fixture_low, node: fixture_node, instrument: fixture_instrument, - context_depth: token_count(count: 400000), concurrent_sessions: 4, - outcome: RunnerServed { - reported_total_buffer: byte_size(count: 28000000000), - reported_gpu_buffer: byte_size(count: 28000000000), - }, - }, + served(b: b_fixture_low, node: fixture_node, instrument: fixture_instrument, + depth: 400000, sessions: 4, resident: 28000000000), ]) is_unobserved_verdict(verdict: evaluate_candidate( candidate: refused_at_four, node: fixture_node, constraints: floor_400k)) @@ -1021,7 +1022,7 @@ fn is_admissible(verdict: CandidateVerdict) -> Bool { // other way. So it must now come back unanswerable. Flip the binding back to `>=` in // slowest_fresh_rate_at_floor and this witness goes red. data busy_only_prefill: List = [ - FreshPrefillObservation { realization: r_iq2_xxs, node: serving_node, concurrent_sessions: 4, depth: token_count(count: 400060), rate: tokens_per_second(count: 90) }, + prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 4, depth: 400060, rate: 90), ] test fn w_a_busier_prefill_reading_does_not_answer_a_quieter_demand() -> Bool { @@ -1057,13 +1058,7 @@ test fn w_a_prefill_reading_at_lower_concurrency_or_another_node_does_not_qualif fn quiet_rate_at(node: NonEmptyStr, hot: Nat) -> TokensPerSecond? { prefill_rate_at_floor( - observations: [FreshPrefillObservation { - realization: r_iq2_xxs, - node: serving_node, - concurrent_sessions: 1, - depth: token_count(count: 400060), - rate: tokens_per_second(count: 135), - }], + observations: [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135)], realization: r_iq2_xxs, node: node, hot_sessions: hot, @@ -1089,14 +1084,19 @@ fn structural_digest(hex: Fnv1a64StructuralDigestHex) -> ContentHash { as_content_hash_structural(structural: structural_content_hash(digest: hex)) } +// The SAME build with the artifact digest recorded in the fnv1a structural family instead of +// sha256. Everything else -- release, quantization, runtime, argv -- is identical, so the only axis +// that cannot be lined up is the digest. +data b_iq2_xxs_structural_artifact: FixtureBuild = FixtureBuild { + release_id: b_iq2_xxs.release_id, + quant: b_iq2_xxs.quant, + artifact: structural_digest(hex: "32af248f4cab44ff" as Fnv1a64StructuralDigestHex), + runtime: b_iq2_xxs.runtime, + argv: b_iq2_xxs.argv, +} + fn r_iq2_xxs_structural_artifact() -> ServingRealizationIdentity { - realization( - release_id: r_iq2_xxs.release_id, - quant: r_iq2_xxs.quant_label, - artifact: structural_digest(hex: "32af248f4cab44ff" as Fnv1a64StructuralDigestHex), - runtime: ollama_runtime, - fixed_mode: serial_mode, - ) + realization_of(b: b_iq2_xxs_structural_artifact) } // THE SCOPE OF THE STALL, which the witness above cannot see because its incomparable reading is @@ -1113,18 +1113,148 @@ fn r_iq2_xxs_structural_artifact() -> ServingRealizationIdentity { // Revert first_incomparable_axis to a node-only filter and this goes red. test fn w_an_irrelevant_cross_family_reading_does_not_stall_the_candidate() -> Bool { let mixed = with_memory_observations(base: build_iq2_xxs, observations: [ - served(r: r_iq2_xxs_structural_artifact(), node: serving_node, instrument: ps_instrument, + served(b: b_iq2_xxs_structural_artifact, node: serving_node, instrument: ps_instrument, depth: 8192, sessions: 1, resident: 80000000000), - served(r: r_iq2_xxs, node: serving_node, instrument: ps_instrument, + served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907), ]) is_admissible(verdict: evaluate_candidate( candidate: mixed, node: serving_node, constraints: floor_400k)) } +// THE SAME DEFECT ON THE OTHER TWO AXES. The memory axis learned to say "held but unalignable"; +// semantic context and prefill were still reading serving_realization_equal, which folds +// RealizationIncomparable to `false`. So a retrieval receipt or a prefill reading whose artifact +// digest was recorded in a different hash family left the population silently, and the candidate +// reported SemanticContextUnverifiedAtFloor or PrefillRateUnmeasuredAtFloor -- telling the operator +// to go measure something that has already been measured. Both must now surface the axis instead. +fn incomparable_axis_of_verdict(verdict: CandidateVerdict) -> NonEmptyStr? { + match verdict { + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => + match m { + EvidenceIdentityIncomparable { axis: a } => Present { value: a } + MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => none + MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => none + MemoryAttemptFailedForNonMemoryCause { cause: _ } => none + SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => none + PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => none + } + ServingCandidateAdmissible { candidate: _, fit: _ } => none + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => none + } +} + +fn refuses_on_artifact_digest(verdict: CandidateVerdict) -> Bool { + match incomparable_axis_of_verdict(verdict: verdict) { + Present { value: a } => (a as String) == "artifact digest" + Absent => false + } +} + +// Memory fit is established by a good reading, so the selector reaches the semantic-context test -- +// where the only receipt on hand belongs to a realization whose digest cannot be lined up. +test fn w_a_cross_family_semantic_receipt_refuses_rather_than_reading_as_unverified() -> Bool { + let c = QuantizedCandidate { + realization: r_iq2_xxs, + quality_rank: 1, + declared_context: token_count(count: 1048576), + semantic_context: verified_at(r: r_iq2_xxs_structural_artifact(), depth: 400060), + fresh_prefill_observations: [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135)], + runtime_memory_observations: [served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907)], + } + refuses_on_artifact_digest(verdict: evaluate_candidate( + candidate: c, node: serving_node, constraints: floor_400k)) +} + +// Memory and semantic context both pass, so the selector reaches prefill -- where the only reading +// at the demanded point belongs to an unalignable realization. +test fn w_a_cross_family_prefill_receipt_refuses_rather_than_reading_as_unmeasured() -> Bool { + let c = QuantizedCandidate { + realization: r_iq2_xxs, + quality_rank: 1, + declared_context: token_count(count: 1048576), + semantic_context: verified_at(r: r_iq2_xxs, depth: 400060), + fresh_prefill_observations: [prefill_of(b: b_iq2_xxs_structural_artifact, node: serving_node, sessions: 1, depth: 400060, rate: 135)], + runtime_memory_observations: [served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907)], + } + refuses_on_artifact_digest(verdict: evaluate_candidate( + candidate: c, node: serving_node, constraints: floor_400k)) +} + +// AN INCOMPARABLE ROW THAT CANNOT CHANGE THE ANSWER MUST NOT STALL IT. A comparable 400k one-session +// success establishes fit. Beside it sits a cross-family 1M one-session success: if it describes +// this candidate it is a second success, and if it describes another realization it is irrelevant. +// Fit holds either way, so the identity question has no bearing and the candidate must be +// admissible. Under the point-relevance-only rule it returned MemoryFitIdentityIncomparable, which +// made a candidate LESS answerable for holding an extra success. +test fn w_a_redundant_incomparable_success_does_not_overturn_established_fit() -> Bool { + let c = with_memory_observations(base: build_iq2_xxs, observations: [ + served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, + depth: 400000, sessions: 1, resident: 87692389907), + served(b: b_iq2_xxs_structural_artifact, node: serving_node, instrument: ps_instrument, + depth: 1048576, sessions: 1, resident: 88865253620), + ]) + is_admissible(verdict: evaluate_candidate( + candidate: c, node: serving_node, constraints: floor_400k)) +} + +// THE OTHER HALF, so the rule above is not read as "ignore incomparable evidence". An incomparable +// EXACT REFUSAL beside a comparable success is still load-bearing: resolving its identity is exactly +// what separates a contradiction from ordinary fit, so it must still stall. +test fn w_an_incomparable_exact_refusal_beside_a_success_still_stalls() -> Bool { + let c = with_memory_observations(base: build_iq2_xxs, observations: [ + served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, + depth: 400000, sessions: 1, resident: 87692389907), + attempt_of(b: b_iq2_xxs_structural_artifact, node: serving_node, instrument: ps_instrument, + depth: 400000, sessions: 1, outcome: out_of_memory), + ]) + refuses_on_artifact_digest(verdict: evaluate_candidate( + candidate: c, node: serving_node, constraints: floor_400k)) +} + +// THE ORDERING TIE THAT SURVIVED THE FIRST REPAIR. Two non-memory failures at the same exact point, +// same realization, host and instrument, differing ONLY in cause. Depth, sessions, footprint and +// instrument are all equal, so the previous key tied and shallowest_of kept whichever the roster +// listed first -- and attempt_failure_cause then reported that cause in the diagnosis. Reversing the +// roster reversed the answer. The reported cause must now be the same in both orders. +fn failure_cause_reported(observations: List) -> NonEmptyStr? { + match evaluate_candidate( + candidate: with_memory_observations(base: build_iq2_xxs, observations: observations), + node: serving_node, constraints: floor_400k) { + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => + match m { + MemoryAttemptFailedForNonMemoryCause { cause: c } => Present { value: c } + MemoryFitUnobservedAtConfiguration { floor: _, sessions: _ } => none + MemoryFitContradictedAtConfiguration { floor: _, sessions: _ } => none + EvidenceIdentityIncomparable { axis: _ } => none + SemanticContextUnverifiedAtFloor { floor: _, declared: _ } => none + PrefillRateUnmeasuredAtFloor { floor: _, regime: _ } => none + } + ServingCandidateAdmissible { candidate: _, fit: _ } => none + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => none + } +} + +test fn w_two_failures_differing_only_in_cause_report_the_same_one_in_both_orders() -> Bool { + let template = attempt_of(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, + depth: 400000, sessions: 1, + outcome: RunnerFailedForOtherCause { cause: "template render failed" as NonEmptyStr }) + let transport = attempt_of(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, + depth: 400000, sessions: 1, + outcome: RunnerFailedForOtherCause { cause: "transport reset" as NonEmptyStr }) + match failure_cause_reported(observations: [template, transport]) { + Absent => false + Present { value: a } => + match failure_cause_reported(observations: [transport, template]) { + Absent => false + Present { value: b } => (a as String) == (b as String) + } + } +} + test fn w_a_cross_family_digest_refuses_rather_than_reading_as_unobserved() -> Bool { let cross_family = with_memory_observations(base: build_iq2_xxs, observations: [ - served(r: r_iq2_xxs_structural_artifact(), node: serving_node, instrument: ps_instrument, + served(b: b_iq2_xxs_structural_artifact, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907), ]) match evaluate_candidate(candidate: cross_family, node: serving_node, constraints: floor_400k) { @@ -1210,6 +1340,11 @@ test fn w_all_serving_choice_claims_hold() -> Bool { && w_a_prefill_reading_at_lower_concurrency_or_another_node_does_not_qualify() && w_a_busier_prefill_reading_does_not_answer_a_quieter_demand() && w_a_cross_family_digest_refuses_rather_than_reading_as_unobserved() + && w_a_cross_family_semantic_receipt_refuses_rather_than_reading_as_unverified() + && w_a_cross_family_prefill_receipt_refuses_rather_than_reading_as_unmeasured() + && w_a_redundant_incomparable_success_does_not_overturn_established_fit() + && w_an_incomparable_exact_refusal_beside_a_success_still_stalls() + && w_two_failures_differing_only_in_cause_report_the_same_one_in_both_orders() && w_an_irrelevant_cross_family_reading_does_not_stall_the_candidate() && w_realization_comparison_reaches_all_three_states_and_different_wins() } @@ -1226,7 +1361,7 @@ data build_unmeasured: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_unmeasured, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_unmeasured, node: serving_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 135) }, + prefill_of(b: b_unmeasured, node: serving_node, sessions: 1, depth: 400060, rate: 135), ], runtime_memory_observations: [], } @@ -1272,10 +1407,10 @@ data other_release: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_qwen, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_qwen, node: serving_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 900) }, + prefill_of(b: b_qwen, node: serving_node, sessions: 1, depth: 400060, rate: 900), ], runtime_memory_observations: [ - served(r: r_qwen, node: serving_node, instrument: fixture_instrument, + served(b: b_qwen, node: serving_node, instrument: fixture_instrument, depth: 1048576, sessions: 1, resident: 30600000000), ], } @@ -1323,9 +1458,9 @@ data fixture_tied_a: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_tied_a, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_tied_a, node: fixture_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + prefill_of(b: b_tied_a, node: fixture_node, sessions: 1, depth: 400060, rate: 500), ], - runtime_memory_observations: [fixture_footprint(r: r_tied_a, resident: 28000000000)], + runtime_memory_observations: [fixture_footprint(b: b_tied_a, resident: 28000000000)], } data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { @@ -1334,9 +1469,9 @@ data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_tied_b, depth: 400060), fresh_prefill_observations: [ - FreshPrefillObservation { realization: r_tied_b, node: fixture_node, concurrent_sessions: 1, depth: token_count(count: 400060), rate: tokens_per_second(count: 500) }, + prefill_of(b: b_tied_b, node: fixture_node, sessions: 1, depth: 400060, rate: 500), ], - runtime_memory_observations: [fixture_footprint(r: r_tied_b, resident: 29000000000)], + runtime_memory_observations: [fixture_footprint(b: b_tied_b, resident: 29000000000)], } fn is_tie_refusal(choice: ServingChoice) -> Bool { From 3adbb46ca76adc04d5afb5a2468ee7f0b4a481e2 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Wed, 2 Sep 2026 02:50:23 +0000 Subject: [PATCH 19/22] wip: review fixes for byte carrier and coproduct folds --- dag/gunbc/model/choice.dag | 72 +++++++++++++++++++++++---------- dag/gunbc/model/population.dag | 27 +++++++++++-- dag/gunbc/model/publication.dag | 23 +++++++++-- 3 files changed, 92 insertions(+), 30 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index d826ee0b7dc..44e214facde 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -996,14 +996,13 @@ fn observation_order_key_before( let ai = a.instrument as String let bi = b.instrument as String if ai != bi { ai < bi } else { - let at = reported_total_bytes(o: a) - let bt = reported_total_bytes(o: b) - if at != bt { at < bt } else { - let ag = reported_gpu_bytes(o: a) - let bg = reported_gpu_bytes(o: b) - if ag != bg { ag < bg } else { - outcome_detail_key(o: a) < outcome_detail_key(o: b) - } + match buffer_axis_order(a: reported_total_buffer(o: a), b: reported_total_buffer(o: b)) { + Present { value: decided } => decided + Absent => + match buffer_axis_order(a: reported_gpu_buffer(o: a), b: reported_gpu_buffer(o: b)) { + Present { value: decided } => decided + Absent => outcome_detail_key(o: a) < outcome_detail_key(o: b) + } } } } @@ -1027,25 +1026,54 @@ fn observation_order_key_before( // only ever breaks ties WITHIN an arm. Two observations equal on this key are equal on every field // the verdict can show, so which one is returned genuinely carries no information -- which is the // property the previous key claimed and did not have. -// The two byte axes compare NUMERICALLY and the prose axis lexicographically, rather than folding -// everything into one string: string order over decimal digits puts "9" after "10", which would be a -// deterministic but wrong magnitude ordering, and the whole point of ordering on footprint is that -// the smaller reading is the closer one. Non-served arms report no buffer and contribute zero, which -// is not a fabricated measurement -- memory_fit_evidence partitions by outcome before any comparison -// happens, so a served and a refused reading are never compared against each other. -fn reported_total_bytes(o: OllamaRunnerMemoryObservation) -> Int { +// THE BYTE AXES STAY ByteSize, AND AN UNREPORTED BUFFER IS ABSENT RATHER THAN ZERO. +// +// The first version of this projected both axes to bare `Int` and gave the non-served arms a literal +// 0. Two defects in one move. The measure carrier was erased at exactly the comparison that cares +// about magnitude, which is where std.measure's authority is load-bearing rather than decorative -- +// and `byte_size_count` yields a Nat, so the Int was a SECOND erasure past the one the authority +// offers. And a refusal reports no buffer at all, so 0 was a fabricated reading: it says the runner +// measured zero bytes, which is a different claim from "the runner did not report a footprint". +// +// So the accessor returns `ByteSize?`. The three arms are already partitioned by +// memory_fit_evidence before anything is ordered, so the Absent case never actually decides a +// comparison here -- but modeling it as absent means nothing has to be fabricated for it to be +// total, and if the partition ever changed the ordering would still be honest. +fn reported_total_buffer(o: OllamaRunnerMemoryObservation) -> ByteSize? { match o.outcome { - RunnerServed { reported_total_buffer: t, reported_gpu_buffer: _ } => byte_size_count(b: t) - RunnerRefusedForMemory { detail: _ } => 0 - RunnerFailedForOtherCause { cause: _ } => 0 + RunnerServed { reported_total_buffer: t, reported_gpu_buffer: _ } => Present { value: t } + RunnerRefusedForMemory { detail: _ } => none + RunnerFailedForOtherCause { cause: _ } => none } } -fn reported_gpu_bytes(o: OllamaRunnerMemoryObservation) -> Int { +fn reported_gpu_buffer(o: OllamaRunnerMemoryObservation) -> ByteSize? { match o.outcome { - RunnerServed { reported_total_buffer: _, reported_gpu_buffer: g } => byte_size_count(b: g) - RunnerRefusedForMemory { detail: _ } => 0 - RunnerFailedForOtherCause { cause: _ } => 0 + RunnerServed { reported_total_buffer: _, reported_gpu_buffer: g } => Present { value: g } + RunnerRefusedForMemory { detail: _ } => none + RunnerFailedForOtherCause { cause: _ } => none + } +} + +// Absent orders before Present, and two absents decide nothing so the next key runs. The magnitude +// comparison is on the Nat the carrier itself yields, never on a scalar minted here. +fn buffer_axis_order(a: ByteSize?, b: ByteSize?) -> Bool? { + match a { + Absent => + match b { + Absent => none + Present { value: _ } => Present { value: true } + } + Present { value: av } => + match b { + Absent => Present { value: false } + Present { value: bv } => + if byte_size_count(b: av) != byte_size_count(b: bv) { + Present { value: byte_size_count(b: av) < byte_size_count(b: bv) } + } else { + none + } + } } } diff --git a/dag/gunbc/model/population.dag b/dag/gunbc/model/population.dag index f661c918363..c929521cc9a 100644 --- a/dag/gunbc/model/population.dag +++ b/dag/gunbc/model/population.dag @@ -308,10 +308,29 @@ fn completeness(population: ModelPopulation, question: CompletenessQuestion) -> // A publisher's catalogue spans exactly that publisher and no other. A single distributor's // catalogue spans a channel rather than a publisher, and an operator roster is whatever a human // remembered -- which is the other half of how the original census went stale. -fn coverage_spans_publisher(coverage: CoverageScope, publisher: NonEmptyStr) -> Bool { +// THE SINGLE ELIMINATOR FOR CoverageScope. The predicate below re-matched all three arms to answer +// one question, which is the same second-representation defect channel_presence_fold closes in +// gunbc.model.publication: a fourth scope would have to be remembered at the type AND at every +// predicate, and a predicate that forgot it would answer false rather than refuse to compile. The +// fold takes the publisher-scoped arm's payload as a function argument because that arm is the only +// one carrying anything a caller can interrogate. +fn coverage_scope_fold( + coverage: CoverageScope, + publisher_catalog: fn(NonEmptyStr) -> T, + single_distributor: T, + operator_roster: T, +) -> T { match coverage { - PublisherReleaseCatalog { publisher: p } => (p as String) == (publisher as String) - SingleDistributorCatalog { channel: _ } => false - OperatorAssertedRoster => false + PublisherReleaseCatalog { publisher: p } => publisher_catalog(p) + SingleDistributorCatalog { channel: _ } => single_distributor + OperatorAssertedRoster => operator_roster } } + +fn coverage_spans_publisher(coverage: CoverageScope, publisher: NonEmptyStr) -> Bool { + coverage_scope_fold( + coverage: coverage, + publisher_catalog: p => (p as String) == (publisher as String), + single_distributor: false, + operator_roster: false) +} diff --git a/dag/gunbc/model/publication.dag b/dag/gunbc/model/publication.dag index cdfffc7b993..39cbd7969cc 100644 --- a/dag/gunbc/model/publication.dag +++ b/dag/gunbc/model/publication.dag @@ -181,13 +181,28 @@ type DistributionObservation { observed_at: Timestamp } -fn observation_reports_presence(observation: DistributionObservation) -> Bool { - match observation.presence { - PresentInChannel { reference: _, packaging: _ } => true - AbsentFromChannel { probed_reference: _ } => false +// THE SINGLE ELIMINATOR FOR ChannelPresence, for the reason runner_attempt_fold exists one module +// over: a predicate that re-matches a coproduct is a second representation of that coproduct's +// shape, so a third arm would have to be remembered here as well as at the type, and the predicate +// that forgot it would silently answer false rather than fail to compile. One catamorphism, every +// predicate a projection through it. It is not a parallel Kind enum -- minting PresenceYes / +// PresenceNo beside the constructors would be a second NAME for one variant set, which is the +// nicknaming section 3 forbids; a fold introduces no vocabulary at all. +fn channel_presence_fold( + presence: ChannelPresence, + present: T, + absent: T, +) -> T { + match presence { + PresentInChannel { reference: _, packaging: _ } => present + AbsentFromChannel { probed_reference: _ } => absent } } +fn observation_reports_presence(observation: DistributionObservation) -> Bool { + channel_presence_fold(presence: observation.presence, present: true, absent: false) +} + // The diagnostic a channel-absent observation is allowed to render. It states the channel, because a // sentence that omits it is the exact sentence that caused the defect this module repairs. fn channel_absence_diagnostic(observation: DistributionObservation) -> String { From edbfc7d6061e6200d7d40b917ff8076056d65bbc Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Wed, 2 Sep 2026 03:07:57 +0000 Subject: [PATCH 20/22] wip: effective configuration replaces argv key; materiality by receipt --- dag/gunbc/model/choice.dag | 425 +++++++++++++++++++++++-------------- 1 file changed, 265 insertions(+), 160 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index 44e214facde..7432b8c288b 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -151,110 +151,87 @@ fn serving_runtime_equal(a: ServingRuntimeIdentity, b: ServingRuntimeIdentity) - content_hash_eq_structural(left: a.materialization, right: b.materialization) } -// TWO CONFIGURATION GRAINS, because one digest cannot answer for both and conflating them was a -// measured inconsistency in this module's own fixture: a single argv digest was carried for readings -// at 131,072, 262,144, 400,000 and 1,048,576 tokens, while the resolved argv encodes the context and -// the slot count through `-c` and `-np`. One exact full-argv digest cannot identify four points. -// -// FIXED MODE is the argv with the VARIED axes removed -- the flags held constant across a -// measurement series (`--flash`, `--load-mode`, the model path). It is what two readings must share -// before one can bound the other, and it is DERIVED from the observed argv rather than authored. -// -// EXACT CONFIGURATION is the whole argv of ONE measured point. It belongs on the observation, not on -// the realization, because it changes per reading by construction. -// -// Both are derived by the functions below from an argv the observer actually read. There is no mint -// that accepts a bare hash: a digest with no argv behind it is the claim-shaped input this repair -// exists to remove. -type FixedRuntimeModeIdentity sole_constructor { - mode_digest: Fnv1a64Structural -} - -fn argv_digest_of(argv: List) -> Fnv1a64Structural { - fold(argv, content_hash_atom(value: "argv" as NonEmptyStr), (acc, word) => - content_hash_combine_structural(left: acc, right: content_hash_atom(value: word))) -} - -// THE SECOND HALF OF THE ARGV SPLIT. FixedRuntimeModeIdentity answers "is this the same runner -// mode", which is what lets a reading taken at one point speak for another. It deliberately drops -// the varied flags, so by construction it CANNOT say which point a reading was taken at. -// -// That leaves a gap the source described and the types did not close: an author could derive a fixed -// mode from one argv and then write any context_depth and concurrent_sessions beside it. The tuple -// (fixed mode, authored depth, authored sessions) does not establish that all three came from one -// observed invocation. So the exact full-argv identity is carried BY THE OBSERVATION, and both it -// and the realization's fixed mode are minted from ONE argv list in one call -- they cannot be -// authored independently. -// -// WHAT THIS STILL DOES NOT DO, named rather than implied: the depth and session values are not -// PARSED out of the argv and checked against the authored ones. std has no string-to-integer -// admission, so there is nothing to parse them with, and inventing a parser inside this module would -// be a second numeric authority. The class therefore sits at mechanically-preventable rather than -// structurally-impossible, and its next-rung trigger is a std integer-admission capability -// sufficient to read a flag value back as a Nat -- at which point the point fields are derived from -// the argv and the mismatched state has no constructor. +// THE CONFIGURATION IS AN ENVIRONMENT PROJECTION, NOT AN ARGV ONE, AND THIS MODULE HAD IT WRONG. +// +// The predecessor hashed a runner's argv, split into a fixed mode and a varied remainder, and the +// fixture that established the fleet's readings wrote that argv as +// `ollama serve --num-ctx 400000 --parallel 1`. Measured on the actual fleet, that launch form does +// not exist. Both serving hosts run bare `ollama serve` and take their entire configuration through +// systemd `Environment=` directives. So every runner on the fleet hashed to the SAME argv, and two +// units differing in context ceiling and slot count -- the two axes this selector exists to reason +// about -- received one identity. The key discriminated nothing on the only fleet it has. +// +// It is replaced rather than widened, because argv was never the right subject: the authority for +// what configures an Ollama server is `extdeps.ollama.server_env`, which already carries the +// variables as typed axes and renders their canonical wire. +// +// TYPED VALUES FIRST, WIRE SECOND. The identity is not parsed out of an observed string -- there is +// no string-to-integer admission in std, and minting one here would be a second numeric authority. +// It runs the other way: the configuration is held as the typed values the authority declares, and +// its identity is the hash of the CANONICAL WIRE those values render to through +// `ollama_context_length_env_assignment` and `ollama_num_parallel_env_assignment`. An observed unit +// is admitted by rendering a candidate and requiring exact equality with what the host reports, so +// the comparison happens in the authority's own spelling and this module never parses anything. +// +// THE PARTITION IS OWNED HERE, NOT SUPPLIED BY THE CALLER. The predecessor took a `varied_flags` +// list as an argument, which let any caller erase a causal axis from the fixed mode and thereby +// authorize evidence transport across it. The closed projection below is the whole partition: the +// context ceiling and the slot count are the varied, orderable axes; the runtime materialization +// carries the fixed ones; the bind address is deployment identity and is deliberately absent, +// because it cannot change fit; and the models root is artifact-resolution provenance, already +// subsumed once the loaded artifact digest is joined. +type OllamaEffectiveConfiguration sole_constructor { + context_ceiling: TokenCount + serving_slots: PositiveSlotCount +} + +fn ollama_effective_configuration( + context_ceiling: TokenCount, + serving_slots: PositiveSlotCount, +) -> OllamaEffectiveConfiguration { + OllamaEffectiveConfiguration { + context_ceiling: context_ceiling, + serving_slots: serving_slots, + } +} + type ExactRuntimeConfigurationIdentity sole_constructor { - argv_digest: Fnv1a64Structural + wire_digest: Fnv1a64Structural } -fn exact_runtime_configuration_from_argv( - argv: List, +fn exact_runtime_configuration_identity( + configuration: OllamaEffectiveConfiguration, ) -> ExactRuntimeConfigurationIdentity { - ExactRuntimeConfigurationIdentity { argv_digest: argv_digest_of(argv: argv) } + ExactRuntimeConfigurationIdentity { + wire_digest: content_hash_combine_structural( + left: content_hash_atom( + value: ollama_context_length_env_assignment( + default_context: configuration.context_ceiling)), + right: content_hash_atom( + value: ollama_num_parallel_env_assignment(slots: configuration.serving_slots))), + } } fn exact_runtime_configuration_equal( a: ExactRuntimeConfigurationIdentity, b: ExactRuntimeConfigurationIdentity, ) -> Bool { - content_hash_eq_structural(left: a.argv_digest, right: b.argv_digest) -} - -// The varied flags AND their values are DROPPED, not blanked, so the mode digest is invariant to -// their values by construction rather than by the caller remembering to pass matching ones. -// -// A LEFT SCAN, NOT AN INDEX LOOKUP. The predecessor walked positions and read argv[i-1] through an -// at_index helper whose out-of-range arm returned `"" as NonEmptyStr` -- a value that is not a -// NonEmptyStr at all, fabricated to keep a total signature. The scan carries the one bit that -// lookup was reconstructing: whether the PREVIOUS word was a varied flag, so its value is the word -// to drop. Nothing can go out of range, so there is no arm to fabricate for. -type ArgvScan { - kept: List - drop_next: Bool -} - -fn argv_scan_step(acc: ArgvScan, word: NonEmptyStr, varied: List) -> ArgvScan { - if acc.drop_next { - ArgvScan { kept: acc.kept, drop_next: false } - } else { - if word_in(roster: varied, word: word) { - ArgvScan { kept: acc.kept, drop_next: true } - } else { - ArgvScan { kept: flat_map([acc.kept, [word]], g => g), drop_next: false } - } - } -} - -fn drop_varied_flags(argv: List, varied: List) -> List { - fold(argv, ArgvScan { kept: [], drop_next: false }, (acc, word) => - argv_scan_step(acc: acc, word: word, varied: varied)).kept + content_hash_eq_structural(left: a.wire_digest, right: b.wire_digest) } -fn word_in(roster: List, word: NonEmptyStr) -> Bool { - length(filter(roster, r => (r as String) == (word as String))) > 0 -} - -fn fixed_runtime_mode_from_argv( - argv: List, - varied_flags: List, -) -> FixedRuntimeModeIdentity { - FixedRuntimeModeIdentity { - mode_digest: argv_digest_of(argv: drop_varied_flags(argv: argv, varied: varied_flags)), - } -} - -fn fixed_runtime_mode_equal(a: FixedRuntimeModeIdentity, b: FixedRuntimeModeIdentity) -> Bool { - content_hash_eq_structural(left: a.mode_digest, right: b.mode_digest) +// THE ADMISSION TEST FOR AN OBSERVED UNIT, and it is a rendering rather than a parse. A host reports +// the two assignment lines it is running; a candidate typed configuration renders its own; equality +// of the canonical wire admits it. A host whose lines do not match any candidate is not coerced into +// one -- it has no configuration identity here, which is the fail-closed answer. +fn observed_configuration_matches( + candidate: OllamaEffectiveConfiguration, + observed_context_assignment: NonEmptyStr, + observed_parallel_assignment: NonEmptyStr, +) -> Bool { + (ollama_context_length_env_assignment(default_context: candidate.context_ceiling) as String) + == (observed_context_assignment as String) + && (ollama_num_parallel_env_assignment(slots: candidate.serving_slots) as String) + == (observed_parallel_assignment as String) } type ServingRealizationIdentity sole_constructor { @@ -262,7 +239,6 @@ type ServingRealizationIdentity sole_constructor { quant_label: NonEmptyStr artifact_digest: ContentHash runtime: ServingRuntimeIdentity - fixed_mode: FixedRuntimeModeIdentity } // The only mint. sole_constructor already refuses a record literal outside this module; this is the @@ -272,14 +248,12 @@ fn serving_realization_identity( quant_label: NonEmptyStr, artifact_digest: ContentHash, runtime: ServingRuntimeIdentity, - fixed_mode: FixedRuntimeModeIdentity, ) -> ServingRealizationIdentity { ServingRealizationIdentity { release_id: release_id, quant_label: quant_label, artifact_digest: artifact_digest, runtime: runtime, - fixed_mode: fixed_mode, } } @@ -349,10 +323,7 @@ fn compare_serving_realization( axis: "artifact digest" as NonEmptyStr, a: a.artifact_digest, b: b.artifact_digest), ), - later: realization_comparison_then( - earlier: compare_bool_axis(same: serving_runtime_equal(a: a.runtime, b: b.runtime)), - later: compare_bool_axis(same: fixed_runtime_mode_equal(a: a.fixed_mode, b: b.fixed_mode)), - ), + later: compare_bool_axis(same: serving_runtime_equal(a: a.runtime, b: b.runtime)), ) } @@ -371,6 +342,30 @@ fn realization_comparison_fold( } } +// THE CONFIGURATION IDENTITY IS CONSUMED HERE, WHICH IS THE POINT OF MINTING IT. A carried key that +// no filter reads changes no answer, and a provenance field that changes no answer is decoration -- +// the same defect, one level in, as carrying an argv digest that every runner on the fleet shared. +// Two readings taken under different context ceilings or different slot counts are readings of +// different deployments: the slot count multiplies KV cache, so a footprint measured at one slot +// count is not a footprint at another, and the ceiling bounds what either could have been asked. +fn observation_configuration_matches( + o: OllamaRunnerMemoryObservation, + configuration: OllamaEffectiveConfiguration, +) -> Bool { + exact_runtime_configuration_equal( + a: exact_runtime_configuration_identity(configuration: o.configuration), + b: exact_runtime_configuration_identity(configuration: configuration)) +} + +fn prefill_configuration_matches( + o: FreshPrefillObservation, + configuration: OllamaEffectiveConfiguration, +) -> Bool { + exact_runtime_configuration_equal( + a: exact_runtime_configuration_identity(configuration: o.configuration), + b: exact_runtime_configuration_identity(configuration: configuration)) +} + fn serving_realization_equal(a: ServingRealizationIdentity, b: ServingRealizationIdentity) -> Bool { realization_comparison_fold( comparison: compare_serving_realization(a: a, b: b), @@ -453,7 +448,7 @@ fn serving_regime_wire(regime: ServingRegime) -> String { // on one node does not describe another. Both now join. type FreshPrefillObservation sole_constructor { realization: ServingRealizationIdentity - configuration: ExactRuntimeConfigurationIdentity + configuration: OllamaEffectiveConfiguration node: HostIdentity concurrent_sessions: Nat depth: TokenCount @@ -465,23 +460,28 @@ fn observed_fresh_prefill( quant_label: NonEmptyStr, artifact_digest: ContentHash, runtime: ServingRuntimeIdentity, - argv: List, - varied_flags: List, + configuration: OllamaEffectiveConfiguration, node: HostIdentity, concurrent_sessions: Nat, depth: TokenCount, rate: TokensPerSecond, -) -> FreshPrefillObservation { - FreshPrefillObservation { - realization: serving_realization_identity( - release_id: release_id, quant_label: quant_label, artifact_digest: artifact_digest, - runtime: runtime, - fixed_mode: fixed_runtime_mode_from_argv(argv: argv, varied_flags: varied_flags)), - configuration: exact_runtime_configuration_from_argv(argv: argv), - node: node, - concurrent_sessions: concurrent_sessions, - depth: depth, - rate: rate, +) -> FreshPrefillObservation? { + if point_within_configuration( + configuration: configuration, + context_depth: depth, + concurrent_sessions: concurrent_sessions) { + Present { value: FreshPrefillObservation { + realization: serving_realization_identity( + release_id: release_id, quant_label: quant_label, artifact_digest: artifact_digest, + runtime: runtime), + configuration: configuration, + node: node, + concurrent_sessions: concurrent_sessions, + depth: depth, + rate: rate, + } } + } else { + none } } @@ -515,9 +515,11 @@ fn fresh_prefill_at_point( node: HostIdentity, hot_sessions: Nat, floor: TokenCount, + configuration: OllamaEffectiveConfiguration, ) -> List { filter(observations, o => host_identity_eq(a: o.node, b: node) + && prefill_configuration_matches(o: o, configuration: configuration) && o.concurrent_sessions == hot_sessions && token_count_value(t: o.depth) >= token_count_value(t: floor)) } @@ -534,24 +536,41 @@ fn first_prefill_incomparable_axis( }) } +// A NUMERIC FOLD IS NOT AN ARM, so "a comparable row exists" is the wrong materiality test here. +// +// The memory axis takes a conservative WITNESS: one qualifying success establishes fit and a second +// cannot unestablish it, so an incomparable success beside a comparable one changes nothing. Prefill +// takes a conservative AGGREGATE -- the SLOWEST qualifying rate -- and an additional row can lower +// it. The falsifier is concrete: a comparable reading at 135 tok/s, an incomparable one at 50, and a +// floor of 100. If the incomparable row is another encoding of this realization the conservative +// rate is 50 and the candidate is REJECTED; if it belongs to another realization the rate is 135 and +// the candidate is ADMITTED. Resolving the identity flips the top-level verdict, so the honest +// answer is that it cannot be resolved. +// +// A sharper rule could admit when including the incomparable row could not move the minimum -- a +// reading slower than every comparable one only matters if it is slow enough to cross the floor. The +// simple rule is chosen deliberately: it is the conservative one, its cost is a refusal rather than +// a wrong answer, and the sharper version needs the floor threaded into an identity check where it +// does not belong. Any relevant incomparable prefill reading stalls. fn fresh_prefill_qualification( observations: List, realization: ServingRealizationIdentity, node: HostIdentity, hot_sessions: Nat, floor: TokenCount, + configuration: OllamaEffectiveConfiguration, ) -> PrefillQualification { - match slowest_fresh_rate_at_floor( - observations: observations, realization: realization, - node: node, hot_sessions: hot_sessions, floor: floor) { - Present { value: r } => PrefillRateQualified { rate: r } + match first_prefill_incomparable_axis( + observations: fresh_prefill_at_point( + observations: observations, node: node, + hot_sessions: hot_sessions, floor: floor, configuration: configuration), + realization: realization) { + Present { value: a } => PrefillIncomparable { axis: a } Absent => - match first_prefill_incomparable_axis( - observations: fresh_prefill_at_point( - observations: observations, node: node, - hot_sessions: hot_sessions, floor: floor), - realization: realization) { - Present { value: a } => PrefillIncomparable { axis: a } + match slowest_fresh_rate_at_floor( + observations: observations, realization: realization, + node: node, hot_sessions: hot_sessions, floor: floor, configuration: configuration) { + Present { value: r } => PrefillRateQualified { rate: r } Absent => PrefillUnmeasured } } @@ -564,6 +583,7 @@ fn prefill_qualification_at_floor( hot_sessions: Nat, floor: TokenCount, regime: ServingRegime, + configuration: OllamaEffectiveConfiguration, ) -> PrefillQualification { match regime { WarmContinuation => PrefillUnmeasured @@ -571,7 +591,7 @@ fn prefill_qualification_at_floor( FreshSessionPrefill => fresh_prefill_qualification( observations: observations, realization: realization, - node: node, hot_sessions: hot_sessions, floor: floor) + node: node, hot_sessions: hot_sessions, floor: floor, configuration: configuration) } } @@ -582,10 +602,11 @@ fn prefill_rate_at_floor( hot_sessions: Nat, floor: TokenCount, regime: ServingRegime, + configuration: OllamaEffectiveConfiguration, ) -> TokensPerSecond? { match prefill_qualification_at_floor( observations: observations, realization: realization, node: node, - hot_sessions: hot_sessions, floor: floor, regime: regime) { + hot_sessions: hot_sessions, floor: floor, regime: regime, configuration: configuration) { PrefillRateQualified { rate: r } => Present { value: r } PrefillIncomparable { axis: _ } => none PrefillUnmeasured => none @@ -622,10 +643,12 @@ fn slowest_fresh_rate_at_floor( node: HostIdentity, hot_sessions: Nat, floor: TokenCount, + configuration: OllamaEffectiveConfiguration, ) -> TokensPerSecond? { let deep = filter(observations, o => serving_realization_equal(a: o.realization, b: realization) && host_identity_eq(a: o.node, b: node) + && prefill_configuration_matches(o: o, configuration: configuration) && o.concurrent_sessions == hot_sessions && token_count_value(t: o.depth) >= token_count_value(t: floor)) fold(deep, none, (worst, o) => @@ -694,7 +717,7 @@ type RunnerAttemptOutcome type OllamaRunnerMemoryObservation sole_constructor { realization: ServingRealizationIdentity - configuration: ExactRuntimeConfigurationIdentity + configuration: OllamaEffectiveConfiguration node: HostIdentity instrument: NonEmptyStr context_depth: TokenCount @@ -702,37 +725,64 @@ type OllamaRunnerMemoryObservation sole_constructor { outcome: RunnerAttemptOutcome } -// ONE ARGV, BOTH GRAINS. The caller hands over the argv the runner was actually launched with and -// the flags it varies; the fixed mode inside the realization and the exact configuration on the -// observation are both derived here, from that one list. There is no surface on which they can -// disagree, because there is no surface on which they are separately supplied. +// A POINT MUST BE ONE ITS CONFIGURATION COULD HAVE PRODUCED, AND THE MINT REFUSES OTHERWISE. +// +// Carrying a configuration identity beside an authored point is provenance without a join: the +// predecessor's constructor took an argv and a depth and a session count and bound them into one +// record, which proved they arrived in one call and proved nothing about whether they described each +// other. A configuration for an 8,192-token ceiling and one slot could carry a reading at 1,048,576 +// tokens and four sessions, and every downstream filter believed it. +// +// The relation that makes them one fact is a BOUND, and it needs no parser to check. A runner +// configured with a context ceiling cannot have served a deeper attempt than that ceiling, and one +// configured with N slots cannot have observed more than N concurrent sessions. Both comparisons are +// on the typed values the authority already declares, so nothing is read back out of a string. +// +// The mint is therefore partial. An out-of-bound tuple has no observation, rather than an +// observation nobody checked -- and because the carrier is sole_constructor, there is no second way +// to build one. fn observed_memory_attempt( release_id: ReleaseIdentity, quant_label: NonEmptyStr, artifact_digest: ContentHash, runtime: ServingRuntimeIdentity, - argv: List, - varied_flags: List, + configuration: OllamaEffectiveConfiguration, node: HostIdentity, instrument: NonEmptyStr, context_depth: TokenCount, concurrent_sessions: Nat, outcome: RunnerAttemptOutcome, -) -> OllamaRunnerMemoryObservation { - OllamaRunnerMemoryObservation { - realization: serving_realization_identity( - release_id: release_id, quant_label: quant_label, artifact_digest: artifact_digest, - runtime: runtime, - fixed_mode: fixed_runtime_mode_from_argv(argv: argv, varied_flags: varied_flags)), - configuration: exact_runtime_configuration_from_argv(argv: argv), - node: node, - instrument: instrument, +) -> OllamaRunnerMemoryObservation? { + if point_within_configuration( + configuration: configuration, context_depth: context_depth, - concurrent_sessions: concurrent_sessions, - outcome: outcome, + concurrent_sessions: concurrent_sessions) { + Present { value: OllamaRunnerMemoryObservation { + realization: serving_realization_identity( + release_id: release_id, quant_label: quant_label, artifact_digest: artifact_digest, + runtime: runtime), + configuration: configuration, + node: node, + instrument: instrument, + context_depth: context_depth, + concurrent_sessions: concurrent_sessions, + outcome: outcome, + } } + } else { + none } } +fn point_within_configuration( + configuration: OllamaEffectiveConfiguration, + context_depth: TokenCount, + concurrent_sessions: Nat, +) -> Bool { + token_count_value(t: context_depth) + <= token_count_value(t: configuration.context_ceiling) + && concurrent_sessions <= positive_slot_count_value(slots: configuration.serving_slots) +} + // WHAT THE QUALIFYING ATTEMPTS SAY, reconciled by TYPE rather than by byte magnitude. // // The previous version folded every qualifying observation to the smallest reported buffer and only @@ -778,10 +828,12 @@ fn served_qualifying_observations( floor: TokenCount, hot_sessions: Nat, node: HostIdentity, + configuration: OllamaEffectiveConfiguration, ) -> List { filter(observations, o => serving_realization_equal(a: o.realization, b: realization) && host_identity_eq(a: o.node, b: node) + && observation_configuration_matches(o: o, configuration: configuration) && token_count_value(t: o.context_depth) >= token_count_value(t: floor) && o.concurrent_sessions >= hot_sessions && outcome_is_served(o: o)) @@ -793,10 +845,12 @@ fn refused_at_exact_point( floor: TokenCount, hot_sessions: Nat, node: HostIdentity, + configuration: OllamaEffectiveConfiguration, ) -> List { filter(observations, o => serving_realization_equal(a: o.realization, b: realization) && host_identity_eq(a: o.node, b: node) + && observation_configuration_matches(o: o, configuration: configuration) && token_count_value(t: o.context_depth) == token_count_value(t: floor) && o.concurrent_sessions == hot_sessions && outcome_is_memory_refusal(o: o)) @@ -810,10 +864,12 @@ fn failed_otherwise_at_exact_point( floor: TokenCount, hot_sessions: Nat, node: HostIdentity, + configuration: OllamaEffectiveConfiguration, ) -> List { filter(observations, o => serving_realization_equal(a: o.realization, b: realization) && host_identity_eq(a: o.node, b: node) + && observation_configuration_matches(o: o, configuration: configuration) && token_count_value(t: o.context_depth) == token_count_value(t: floor) && o.concurrent_sessions == hot_sessions && outcome_is_other_failure(o: o)) @@ -926,38 +982,82 @@ fn first_axis_in( }) } +// ARM PRESENCE IS NOT THE WHOLE MATERIALITY TEST -- THE CARRIED RECEIPT COUNTS TOO. +// +// The first version of this rule asked only whether a comparable row already occupied the arm, on +// the reasoning that a second success cannot unestablish fit. True for the ARM, false for the +// RECEIPT the arm carries. Two exact-point non-memory failures, one comparable with cause "z" and +// one incomparable with cause "a": the arm is occupied either way, so the old rule suppressed the +// incomparable row and reported "z". But the canonical order is total over the payload, so if that +// row is the same realization it SORTS FIRST and the answer is "a". Identity resolution changes +// MemoryAttemptFailedForNonMemoryCause, which is verdict-visible data -- exactly the class the total +// ordering was introduced to stabilize, discarded before the ordering could see it. +// +// So a relevant incomparable row is load-bearing when either could hold: +// it would INTRODUCE an arm that is otherwise empty, or +// it would WIN the canonical order against the comparable row currently carrying that arm. +// Anything else genuinely cannot move the answer and does not stall. +fn incomparable_rows_in( + rows: List, + realization: ServingRealizationIdentity, +) -> List { + filter(rows, o => + match incomparable_axis_of(o: o, realization: realization) { + Present { value: _ } => true + Absent => false + }) +} + +fn arm_deciding_axis( + arm_rows: List, + comparable_rows: List, + realization: ServingRealizationIdentity, +) -> NonEmptyStr? { + let incomparable = incomparable_rows_in(rows: arm_rows, realization: realization) + match shallowest_of(observations: comparable_rows) { + Absent => first_axis_in(observations: incomparable, realization: realization) + Present { value: carrier } => + first_axis_in( + observations: filter(incomparable, o => + observation_order_key_before(a: o, b: carrier)), + realization: realization) + } +} + fn deciding_incomparable_axis( observations: List, realization: ServingRealizationIdentity, floor: TokenCount, hot_sessions: Nat, node: HostIdentity, + configuration: OllamaEffectiveConfiguration, ) -> NonEmptyStr? { let relevant = filter(observations, o => host_identity_eq(a: o.node, b: node) + && observation_configuration_matches(o: o, configuration: configuration) && participates_at_point(o: o, floor: floor, hot_sessions: hot_sessions)) let served_rows = filter(relevant, o => outcome_is_served(o: o)) let refused_rows = filter(relevant, o => outcome_is_memory_refusal(o: o)) let broken_rows = filter(relevant, o => outcome_is_other_failure(o: o)) let comparable_served = served_qualifying_observations( observations: observations, realization: realization, - floor: floor, hot_sessions: hot_sessions, node: node) + floor: floor, hot_sessions: hot_sessions, node: node, configuration: configuration) let comparable_refused = refused_at_exact_point( observations: observations, realization: realization, - floor: floor, hot_sessions: hot_sessions, node: node) + floor: floor, hot_sessions: hot_sessions, node: node, configuration: configuration) let comparable_broken = failed_otherwise_at_exact_point( observations: observations, realization: realization, - floor: floor, hot_sessions: hot_sessions, node: node) - let from_served = - if length(comparable_served) > 0 { none } - else { first_axis_in(observations: served_rows, realization: realization) } - let from_refused = - if length(comparable_refused) > 0 { none } - else { first_axis_in(observations: refused_rows, realization: realization) } + floor: floor, hot_sessions: hot_sessions, node: node, configuration: configuration) + let from_served = arm_deciding_axis( + arm_rows: served_rows, comparable_rows: comparable_served, realization: realization) + let from_refused = arm_deciding_axis( + arm_rows: refused_rows, comparable_rows: comparable_refused, realization: realization) let from_broken = - if length(comparable_served) > 0 || length(comparable_refused) > 0 - || length(comparable_broken) > 0 { none } - else { first_axis_in(observations: broken_rows, realization: realization) } + if length(comparable_served) > 0 || length(comparable_refused) > 0 { none } + else { + arm_deciding_axis( + arm_rows: broken_rows, comparable_rows: comparable_broken, realization: realization) + } match from_served { Present { value: a } => Present { value: a } Absent => @@ -1111,14 +1211,15 @@ fn memory_fit_evidence( floor: TokenCount, hot_sessions: Nat, node: HostIdentity, + configuration: OllamaEffectiveConfiguration, ) -> MemoryFitEvidence { match deciding_incomparable_axis( observations: observations, realization: realization, - floor: floor, hot_sessions: hot_sessions, node: node) { + floor: floor, hot_sessions: hot_sessions, node: node, configuration: configuration) { Present { value: axis } => MemoryFitIdentityIncomparable { axis: axis } Absent => decidable_memory_fit_evidence( observations: observations, realization: realization, - floor: floor, hot_sessions: hot_sessions, node: node) + floor: floor, hot_sessions: hot_sessions, node: node, configuration: configuration) } } @@ -1128,16 +1229,17 @@ fn decidable_memory_fit_evidence( floor: TokenCount, hot_sessions: Nat, node: HostIdentity, + configuration: OllamaEffectiveConfiguration, ) -> MemoryFitEvidence { let served = served_qualifying_observations( observations: observations, realization: realization, - floor: floor, hot_sessions: hot_sessions, node: node) + floor: floor, hot_sessions: hot_sessions, node: node, configuration: configuration) let refused = refused_at_exact_point( observations: observations, realization: realization, - floor: floor, hot_sessions: hot_sessions, node: node) + floor: floor, hot_sessions: hot_sessions, node: node, configuration: configuration) let broken = failed_otherwise_at_exact_point( observations: observations, realization: realization, - floor: floor, hot_sessions: hot_sessions, node: node) + floor: floor, hot_sessions: hot_sessions, node: node, configuration: configuration) match shallowest_of(observations: served) { Present { value: s } => match shallowest_of(observations: refused) { @@ -1165,6 +1267,7 @@ type ServingConstraints { hot_sessions: Nat prefill_floor: TokensPerSecond prefill_regime: ServingRegime + configuration: OllamaEffectiveConfiguration } // ============================ THE DECISION ============================ @@ -1302,6 +1405,7 @@ fn evaluate_candidate( floor: constraints.context_floor, hot_sessions: constraints.hot_sessions, node: node, + configuration: constraints.configuration, ) { MemoryFitIdentityIncomparable { axis: a } => unanswerable(candidate: candidate, missing: EvidenceIdentityIncomparable { axis: a }) @@ -1345,6 +1449,7 @@ fn evaluate_candidate( hot_sessions: constraints.hot_sessions, floor: constraints.context_floor, regime: constraints.prefill_regime, + configuration: constraints.configuration, ) { PrefillIncomparable { axis: a } => unanswerable(candidate: candidate, missing: EvidenceIdentityIncomparable { axis: a }) From bd09e281d8339fec6c7f5b08050c42659863217b Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Wed, 2 Sep 2026 04:12:15 +0000 Subject: [PATCH 21/22] A launch bound is not an execution receipt, so the point is derived rather than authored OllamaEffectiveConfiguration overstated what two env assignments prove. OLLAMA_CONTEXT_LENGTH is the server's DEFAULT context, not a ceiling -- a model option or a request num_ctx may replace it -- and OLLAMA_NUM_PARALLEL is the configured slot count, not the number of requests in flight while somebody measured. The bound check admitted an authored (1M tokens, 4 sessions) point for a measurement that was a short prompt on one request, and a one-request success then masqueraded as a four-session success. So the authored point is deleted as an input. An observation's point is derived from an ExecutedRequestReceipt -- the runtime's own prompt_eval_count and the concurrency a named instrument established -- bounded by the effective context, which the override arm may raise past the server default. A point the attempt did not exercise now has no constructor. Also: the two side-chat falsifiers land as discriminating REDs with positive controls, and is_same/is_different/is_incomparable route through realization_comparison_fold instead of re-matching the coproduct. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01U397y4s3dSBof7vGPAX87G --- dag/gunbc/model/choice.dag | 219 +++++--- .../model/serving_choice_witness_test.dag | 477 ++++++++++++------ 2 files changed, 471 insertions(+), 225 deletions(-) diff --git a/dag/gunbc/model/choice.dag b/dag/gunbc/model/choice.dag index 7432b8c288b..181662b311a 100644 --- a/dag/gunbc/model/choice.dag +++ b/dag/gunbc/model/choice.dag @@ -19,6 +19,10 @@ import std.measure { ByteSize, byte_size, byte_size_count, TokenCount, token_count, token_count_value, TokensPerSecond, tokens_per_second_count, + PositiveSlotCount, positive_slot_count_value, +} +import extdeps.ollama.server_env { + ollama_context_length_env_assignment, ollama_num_parallel_env_assignment, } // CHOOSING WHAT TO SERVE, as a function rather than as an argument. @@ -180,17 +184,17 @@ fn serving_runtime_equal(a: ServingRuntimeIdentity, b: ServingRuntimeIdentity) - // carries the fixed ones; the bind address is deployment identity and is deliberately absent, // because it cannot change fit; and the models root is artifact-resolution provenance, already // subsumed once the loaded artifact digest is joined. -type OllamaEffectiveConfiguration sole_constructor { - context_ceiling: TokenCount +type ObservedOllamaLaunchConfiguration sole_constructor { + server_default_context: TokenCount serving_slots: PositiveSlotCount } -fn ollama_effective_configuration( - context_ceiling: TokenCount, +fn observed_ollama_launch_configuration( + server_default_context: TokenCount, serving_slots: PositiveSlotCount, -) -> OllamaEffectiveConfiguration { - OllamaEffectiveConfiguration { - context_ceiling: context_ceiling, +) -> ObservedOllamaLaunchConfiguration { + ObservedOllamaLaunchConfiguration { + server_default_context: server_default_context, serving_slots: serving_slots, } } @@ -200,13 +204,13 @@ type ExactRuntimeConfigurationIdentity sole_constructor { } fn exact_runtime_configuration_identity( - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> ExactRuntimeConfigurationIdentity { ExactRuntimeConfigurationIdentity { wire_digest: content_hash_combine_structural( left: content_hash_atom( value: ollama_context_length_env_assignment( - default_context: configuration.context_ceiling)), + default_context: configuration.server_default_context)), right: content_hash_atom( value: ollama_num_parallel_env_assignment(slots: configuration.serving_slots))), } @@ -224,11 +228,11 @@ fn exact_runtime_configuration_equal( // of the canonical wire admits it. A host whose lines do not match any candidate is not coerced into // one -- it has no configuration identity here, which is the fail-closed answer. fn observed_configuration_matches( - candidate: OllamaEffectiveConfiguration, + candidate: ObservedOllamaLaunchConfiguration, observed_context_assignment: NonEmptyStr, observed_parallel_assignment: NonEmptyStr, ) -> Bool { - (ollama_context_length_env_assignment(default_context: candidate.context_ceiling) as String) + (ollama_context_length_env_assignment(default_context: candidate.server_default_context) as String) == (observed_context_assignment as String) && (ollama_num_parallel_env_assignment(slots: candidate.serving_slots) as String) == (observed_parallel_assignment as String) @@ -350,7 +354,7 @@ fn realization_comparison_fold( // count is not a footprint at another, and the ceiling bounds what either could have been asked. fn observation_configuration_matches( o: OllamaRunnerMemoryObservation, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> Bool { exact_runtime_configuration_equal( a: exact_runtime_configuration_identity(configuration: o.configuration), @@ -359,7 +363,7 @@ fn observation_configuration_matches( fn prefill_configuration_matches( o: FreshPrefillObservation, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> Bool { exact_runtime_configuration_equal( a: exact_runtime_configuration_identity(configuration: o.configuration), @@ -448,7 +452,7 @@ fn serving_regime_wire(regime: ServingRegime) -> String { // on one node does not describe another. Both now join. type FreshPrefillObservation sole_constructor { realization: ServingRealizationIdentity - configuration: OllamaEffectiveConfiguration + configuration: ObservedOllamaLaunchConfiguration node: HostIdentity concurrent_sessions: Nat depth: TokenCount @@ -460,28 +464,24 @@ fn observed_fresh_prefill( quant_label: NonEmptyStr, artifact_digest: ContentHash, runtime: ServingRuntimeIdentity, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, node: HostIdentity, - concurrent_sessions: Nat, - depth: TokenCount, + receipt: ExecutedRequestReceipt, rate: TokensPerSecond, ) -> FreshPrefillObservation? { - if point_within_configuration( - configuration: configuration, - context_depth: depth, - concurrent_sessions: concurrent_sessions) { - Present { value: FreshPrefillObservation { - realization: serving_realization_identity( - release_id: release_id, quant_label: quant_label, artifact_digest: artifact_digest, - runtime: runtime), - configuration: configuration, - node: node, - concurrent_sessions: concurrent_sessions, - depth: depth, - rate: rate, - } } - } else { - none + match exercised_attempt_point(configuration: configuration, receipt: receipt) { + Absent => none + Present { value: point } => + Present { value: FreshPrefillObservation { + realization: serving_realization_identity( + release_id: release_id, quant_label: quant_label, artifact_digest: artifact_digest, + runtime: runtime), + configuration: configuration, + node: node, + concurrent_sessions: point.concurrent_sessions, + depth: point.context_depth, + rate: rate, + } } } } @@ -515,7 +515,7 @@ fn fresh_prefill_at_point( node: HostIdentity, hot_sessions: Nat, floor: TokenCount, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> List { filter(observations, o => host_identity_eq(a: o.node, b: node) @@ -558,7 +558,7 @@ fn fresh_prefill_qualification( node: HostIdentity, hot_sessions: Nat, floor: TokenCount, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> PrefillQualification { match first_prefill_incomparable_axis( observations: fresh_prefill_at_point( @@ -583,7 +583,7 @@ fn prefill_qualification_at_floor( hot_sessions: Nat, floor: TokenCount, regime: ServingRegime, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> PrefillQualification { match regime { WarmContinuation => PrefillUnmeasured @@ -602,7 +602,7 @@ fn prefill_rate_at_floor( hot_sessions: Nat, floor: TokenCount, regime: ServingRegime, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> TokensPerSecond? { match prefill_qualification_at_floor( observations: observations, realization: realization, node: node, @@ -643,7 +643,7 @@ fn slowest_fresh_rate_at_floor( node: HostIdentity, hot_sessions: Nat, floor: TokenCount, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> TokensPerSecond? { let deep = filter(observations, o => serving_realization_equal(a: o.realization, b: realization) @@ -717,7 +717,7 @@ type RunnerAttemptOutcome type OllamaRunnerMemoryObservation sole_constructor { realization: ServingRealizationIdentity - configuration: OllamaEffectiveConfiguration + configuration: ObservedOllamaLaunchConfiguration node: HostIdentity instrument: NonEmptyStr context_depth: TokenCount @@ -746,43 +746,116 @@ fn observed_memory_attempt( quant_label: NonEmptyStr, artifact_digest: ContentHash, runtime: ServingRuntimeIdentity, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, node: HostIdentity, instrument: NonEmptyStr, - context_depth: TokenCount, - concurrent_sessions: Nat, + receipt: ExecutedRequestReceipt, outcome: RunnerAttemptOutcome, ) -> OllamaRunnerMemoryObservation? { - if point_within_configuration( - configuration: configuration, - context_depth: context_depth, - concurrent_sessions: concurrent_sessions) { - Present { value: OllamaRunnerMemoryObservation { - realization: serving_realization_identity( - release_id: release_id, quant_label: quant_label, artifact_digest: artifact_digest, - runtime: runtime), - configuration: configuration, - node: node, - instrument: instrument, - context_depth: context_depth, - concurrent_sessions: concurrent_sessions, - outcome: outcome, + match exercised_attempt_point(configuration: configuration, receipt: receipt) { + Absent => none + Present { value: point } => + Present { value: OllamaRunnerMemoryObservation { + realization: serving_realization_identity( + release_id: release_id, quant_label: quant_label, artifact_digest: artifact_digest, + runtime: runtime), + configuration: configuration, + node: node, + instrument: instrument, + context_depth: point.context_depth, + concurrent_sessions: point.concurrent_sessions, + outcome: outcome, + } } + } +} + +// A BOUND IS NOT A RECEIPT, AND THE NAME NOW SAYS SO. `OLLAMA_CONTEXT_LENGTH` is the server's +// DEFAULT context, not a ceiling: a model's own options or a request's `num_ctx` may replace it. +// `OLLAMA_NUM_PARALLEL` is the CONFIGURED slot count, not the number of requests that were actually +// in flight while somebody measured. So the launch wires establish what a point is PERMITTED to be; +// they establish nothing about what any particular attempt exercised. +// +// The gap that closes here is exact: under a 1M / 4-slot unit the predecessor's bound check admitted +// an authored (1,000,000 tokens, 4 sessions) point for a measurement that was really a short prompt +// on one request. Both inequalities pass, and a one-request success then masquerades as a +// four-session success and generalizes downward through the memory rule. +// +// So the authored point is DELETED as an input rather than checked. An observation's point is +// derived from an execution receipt -- the prompt token count the runtime reported back, and the +// concurrency a named instrument established -- and the bound is applied to that receipt. A caller +// cannot state a point at all, so a point the attempt did not exercise has no constructor. +type RequestContextOverrideEvidence + = NoModelOrRequestContextOverride + | RequestContextOverrideObserved { effective_context: TokenCount } + +// The effective context for one attempt: the server default unless the model or the request replaced +// it, in which case the replacement is the bound. This is why the launch field may not be called a +// ceiling -- the override may be LARGER. +fn effective_context_for_attempt( + configuration: ObservedOllamaLaunchConfiguration, + override: RequestContextOverrideEvidence, +) -> TokenCount { + match override { + NoModelOrRequestContextOverride => configuration.server_default_context + RequestContextOverrideObserved { effective_context: c } => c + } +} + +// WHAT AN EXECUTION ACTUALLY REPORTED BACK. `prompt_eval_count` is the runtime's own count of the +// tokens it evaluated -- the fleet's 308,005-token reading is one of these -- and +// `concurrent_requests` is what the named instrument established was in flight, never what the slot +// count permits. +type ExecutedRequestReceipt sole_constructor { + prompt_eval_count: Nat + concurrent_requests: Nat + instrument: NonEmptyStr + context_override: RequestContextOverrideEvidence +} + +fn executed_request_receipt( + prompt_eval_count: Nat, + concurrent_requests: Nat, + instrument: NonEmptyStr, + context_override: RequestContextOverrideEvidence, +) -> ExecutedRequestReceipt { + ExecutedRequestReceipt { + prompt_eval_count: prompt_eval_count, + concurrent_requests: concurrent_requests, + instrument: instrument, + context_override: context_override, + } +} + +type ExercisedAttemptPoint sole_constructor { + context_depth: TokenCount + concurrent_sessions: Nat + instrument: NonEmptyStr +} + +// The mint stays partial, and the partiality now means something stronger: a receipt reporting more +// evaluated tokens than the effective context, more in-flight requests than the unit has slots, or +// zero requests, describes an execution this configuration could not have performed -- so the +// instrument disagrees with the unit and the reading is refused rather than reconciled. +fn exercised_attempt_point( + configuration: ObservedOllamaLaunchConfiguration, + receipt: ExecutedRequestReceipt, +) -> ExercisedAttemptPoint? { + if receipt.prompt_eval_count + <= token_count_value(t: effective_context_for_attempt( + configuration: configuration, override: receipt.context_override)) + && receipt.concurrent_requests + <= positive_slot_count_value(slots: configuration.serving_slots) + && receipt.concurrent_requests >= 1 { + Present { value: ExercisedAttemptPoint { + context_depth: token_count(count: receipt.prompt_eval_count), + concurrent_sessions: receipt.concurrent_requests, + instrument: receipt.instrument, } } } else { none } } -fn point_within_configuration( - configuration: OllamaEffectiveConfiguration, - context_depth: TokenCount, - concurrent_sessions: Nat, -) -> Bool { - token_count_value(t: context_depth) - <= token_count_value(t: configuration.context_ceiling) - && concurrent_sessions <= positive_slot_count_value(slots: configuration.serving_slots) -} - // WHAT THE QUALIFYING ATTEMPTS SAY, reconciled by TYPE rather than by byte magnitude. // // The previous version folded every qualifying observation to the smallest reported buffer and only @@ -828,7 +901,7 @@ fn served_qualifying_observations( floor: TokenCount, hot_sessions: Nat, node: HostIdentity, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> List { filter(observations, o => serving_realization_equal(a: o.realization, b: realization) @@ -845,7 +918,7 @@ fn refused_at_exact_point( floor: TokenCount, hot_sessions: Nat, node: HostIdentity, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> List { filter(observations, o => serving_realization_equal(a: o.realization, b: realization) @@ -864,7 +937,7 @@ fn failed_otherwise_at_exact_point( floor: TokenCount, hot_sessions: Nat, node: HostIdentity, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> List { filter(observations, o => serving_realization_equal(a: o.realization, b: realization) @@ -1030,7 +1103,7 @@ fn deciding_incomparable_axis( floor: TokenCount, hot_sessions: Nat, node: HostIdentity, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> NonEmptyStr? { let relevant = filter(observations, o => host_identity_eq(a: o.node, b: node) @@ -1211,7 +1284,7 @@ fn memory_fit_evidence( floor: TokenCount, hot_sessions: Nat, node: HostIdentity, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> MemoryFitEvidence { match deciding_incomparable_axis( observations: observations, realization: realization, @@ -1229,7 +1302,7 @@ fn decidable_memory_fit_evidence( floor: TokenCount, hot_sessions: Nat, node: HostIdentity, - configuration: OllamaEffectiveConfiguration, + configuration: ObservedOllamaLaunchConfiguration, ) -> MemoryFitEvidence { let served = served_qualifying_observations( observations: observations, realization: realization, @@ -1267,7 +1340,7 @@ type ServingConstraints { hot_sessions: Nat prefill_floor: TokensPerSecond prefill_regime: ServingRegime - configuration: OllamaEffectiveConfiguration + configuration: ObservedOllamaLaunchConfiguration } // ============================ THE DECISION ============================ diff --git a/dag/test/claim/model/serving_choice_witness_test.dag b/dag/test/claim/model/serving_choice_witness_test.dag index 94dac4843a7..559859a0d82 100644 --- a/dag/test/claim/model/serving_choice_witness_test.dag +++ b/dag/test/claim/model/serving_choice_witness_test.dag @@ -21,7 +21,11 @@ import gunbc.model.choice { ServingRealizationIdentity, serving_realization_equal, serving_realization_identity, ServingRuntimeIdentity, serving_runtime_identity_for_release, serving_runtime_identity_from_observation, - FixedRuntimeModeIdentity, fixed_runtime_mode_from_argv, + ObservedOllamaLaunchConfiguration, observed_ollama_launch_configuration, + ExactRuntimeConfigurationIdentity, exact_runtime_configuration_identity, + observed_memory_attempt, observed_fresh_prefill, + ExecutedRequestReceipt, executed_request_receipt, NoModelOrRequestContextOverride, + RequestContextOverrideObserved, SemanticContextEvidence, OllamaRunnerMemoryObservation, memory_fit_evidence, RunnerAttemptOutcome, RunnerServed, RunnerRefusedForMemory, RunnerFailedForOtherCause, @@ -31,7 +35,7 @@ import gunbc.model.choice { EvidenceIdentityIncomparable, MemoryFitIdentityIncomparable, outcome_is_memory_refusal, DoesNotFitMemory, ContextBelowFloor, PrefillBelowFloor, RealizationComparison, RealizationSame, RealizationDifferent, RealizationIncomparable, - compare_serving_realization, + compare_serving_realization, realization_comparison_fold, FreshSessionPrefill, WarmContinuation, serving_regime_wire, QuantizedCandidate, ServingConstraints, CandidateVerdict, ServingCandidateAdmissible, ServingCandidateRejected, @@ -76,33 +80,39 @@ data fixture_node: NonEmptyStr = "fixture-node" as NonEmptyStr // `size_vram` agreed exactly on this unified-memory node, so nothing turns on which field is read. data ps_instrument: NonEmptyStr = "/api/ps size and size_vram after load at explicit num_ctx" as NonEmptyStr -// THE FIXED RUNTIME MODE, hashed from the resolved argv with the VARIED flags dropped. The readings -// below were taken with OLLAMA_NUM_PARALLEL unset and num_ctx set per reading; those two are the -// axes the selector varies and asks about, so hashing their values into the mode key would make -// every reading answer for exactly one point and nothing else. They are dropped, not blanked, so the -// key is invariant to them by construction. Everything else in the argv IS in the key: a runner -// launched with different flash-attention or KV-cache-type settings is a different mode, and its -// footprint readings correctly stop answering for this one. -data varied_flags: List = [ - "--num-ctx" as NonEmptyStr, - "--parallel" as NonEmptyStr, -] - -data serial_mode: FixedRuntimeModeIdentity = fixed_runtime_mode_from_argv( - argv: [ - "ollama" as NonEmptyStr, "serve" as NonEmptyStr, - "--num-ctx" as NonEmptyStr, "400000" as NonEmptyStr, - "--parallel" as NonEmptyStr, "1" as NonEmptyStr, - ], - varied_flags: varied_flags, -) - // THE RUNTIME, from the CITED pinned release rather than a name-and-version pair the witness types. // `serving_runtime_identity_for_release` is total because the release row is a closed authority; the // observation-side mint beside it is the one that can come back Absent, and w_an_unknown_runtime_ // asset_has_no_identity below holds it to that. data ollama_runtime: ServingRuntimeIdentity = serving_runtime_identity_for_release(rel: ollama_v0_32_9_release) +// THE MINTS ARE PARTIAL NOW, SO THE FIXTURE HAS TO SAY WHAT AN UNADMITTED READING MEANS. +// +// observed_memory_attempt and observed_fresh_prefill refuse a point their configuration could not +// have produced, so every fixture reading arrives as an option. Filtering the Absent ones away +// silently is exactly the absorbing fallback this module refuses elsewhere: the witness would then +// assert over a SHORTER roster than it wrote, and a fixture that stopped being admitted would +// weaken every claim above it without failing anything. +// +// So the filter exists, and beside it sits a count check. w_every_fixture_reading_was_admitted holds +// each declared roster to its own length, so a reading that stops being admitted turns a witness red +// instead of quietly leaving the population. +fn present_memory(xs: List) -> List { + flat_map(xs, x => + match x { + Present { value: v } => [v] + Absent => [] + }) +} + +fn present_prefill(xs: List) -> List { + flat_map(xs, x => + match x { + Present { value: v } => [v] + Absent => [] + }) +} + // A DIGEST, NOT A TAG, AND A MALFORMED ONE IS UNWRITABLE. Every hex string below is admitted as // Sha256DigestHex, a `String where lower_hex_64` refinement the substrate checks at TYPECHECK for a // literal -- measured: `"zz" as Sha256DigestHex` is a resolve-time type mismatch, not a runtime @@ -119,14 +129,12 @@ fn realization( quant: NonEmptyStr, artifact: ContentHash, runtime: ServingRuntimeIdentity, - fixed_mode: FixedRuntimeModeIdentity, ) -> ServingRealizationIdentity { serving_realization_identity( release_id: release_id, quant_label: quant, artifact_digest: artifact, runtime: runtime, - fixed_mode: fixed_mode, ) } @@ -137,7 +145,7 @@ fn served( depth: Nat, sessions: Nat, resident: Nat, -) -> OllamaRunnerMemoryObservation { +) -> OllamaRunnerMemoryObservation? { attempt_of(b: b, node: node, instrument: instrument, depth: depth, sessions: sessions, outcome: RunnerServed { reported_total_buffer: byte_size(count: resident), @@ -145,7 +153,7 @@ fn served( }) } -fn iq2_xxs_footprint(depth: Nat, resident: Nat) -> OllamaRunnerMemoryObservation { +fn iq2_xxs_footprint(depth: Nat, resident: Nat) -> OllamaRunnerMemoryObservation? { served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: depth, sessions: 1, resident: resident) } @@ -181,16 +189,20 @@ type FixtureBuild { quant: NonEmptyStr artifact: ContentHash runtime: ServingRuntimeIdentity - argv: List + configuration: ObservedOllamaLaunchConfiguration } fn realization_of(b: FixtureBuild) -> ServingRealizationIdentity { serving_realization_identity( release_id: b.release_id, quant_label: b.quant, artifact_digest: b.artifact, - runtime: b.runtime, - fixed_mode: fixed_runtime_mode_from_argv(argv: b.argv, varied_flags: varied_flags)) + runtime: b.runtime) } +// THE MINT IS PARTIAL, so the fixture has to say what an out-of-bound point means. It means the +// fixture is wrong, not that the selector should see something: an unadmitted point is not an +// observation, and a witness that silently dropped one would be asserting over a shorter roster than +// it wrote. So this projects Absent onto a reading that cannot exist, and +// w_a_point_its_configuration_could_not_have_produced_has_no_observation holds the refusal directly. fn attempt_of( b: FixtureBuild, node: HostIdentity, @@ -198,12 +210,24 @@ fn attempt_of( depth: Nat, sessions: Nat, outcome: RunnerAttemptOutcome, -) -> OllamaRunnerMemoryObservation { +) -> OllamaRunnerMemoryObservation? { observed_memory_attempt( release_id: b.release_id, quant_label: b.quant, artifact_digest: b.artifact, - runtime: b.runtime, argv: b.argv, varied_flags: varied_flags, + runtime: b.runtime, configuration: b.configuration, node: node, instrument: instrument, - context_depth: token_count(count: depth), concurrent_sessions: sessions, outcome: outcome) + receipt: fixture_receipt(depth: depth, sessions: sessions, instrument: instrument), + outcome: outcome) +} + +// EVERY FIXTURE POINT IS A RECEIPT, because the constructor no longer accepts an authored one. The +// depth is what a runtime reported as `prompt_eval_count` and the session count is what the named +// instrument established was in flight; the fixtures below carry the fleet's real readings in those +// positions. `NoModelOrRequestContextOverride` is the narrow lane: these readings were taken against +// the server default with no request-level `num_ctx`. +fn fixture_receipt(depth: Nat, sessions: Nat, instrument: NonEmptyStr) -> ExecutedRequestReceipt { + executed_request_receipt( + prompt_eval_count: depth, concurrent_requests: sessions, instrument: instrument, + context_override: NoModelOrRequestContextOverride) } fn prefill_of( @@ -212,35 +236,51 @@ fn prefill_of( sessions: Nat, depth: Nat, rate: Nat, -) -> FreshPrefillObservation { +) -> FreshPrefillObservation? { observed_fresh_prefill( release_id: b.release_id, quant_label: b.quant, artifact_digest: b.artifact, - runtime: b.runtime, argv: b.argv, varied_flags: varied_flags, - node: node, concurrent_sessions: sessions, - depth: token_count(count: depth), rate: tokens_per_second(count: rate)) + runtime: b.runtime, configuration: b.configuration, + node: node, + receipt: fixture_receipt(depth: depth, sessions: sessions, instrument: prefill_instrument), + rate: tokens_per_second(count: rate)) } -data serving_argv: List = [ - "ollama" as NonEmptyStr, "serve" as NonEmptyStr, - "--num-ctx" as NonEmptyStr, "400000" as NonEmptyStr, - "--parallel" as NonEmptyStr, "1" as NonEmptyStr, -] +data prefill_instrument: NonEmptyStr = "ollama /api/generate prompt_eval_count and duration" as NonEmptyStr + + +// THE FLEET'S ACTUAL CONFIGURATION, and it is an environment projection because that is what the +// hosts run. Read off the live serving unit: OLLAMA_CONTEXT_LENGTH=1048576 and +// OLLAMA_NUM_PARALLEL=4, with a bare `ollama serve` argv. The predecessor fixture wrote +// `ollama serve --num-ctx 400000 --parallel 1`, a launch form that exists nowhere on this fleet. +data serving_configuration: ObservedOllamaLaunchConfiguration = observed_ollama_launch_configuration( + server_default_context: token_count(count: 1048576), + serving_slots: positive_slot_count(count: positive_measure_count(predecessor: 3)), +) -data fixture_argv: List = [ - "fixture-runner" as NonEmptyStr, "--fixture" as NonEmptyStr, -] +// A DIFFERENT DEPLOYMENT, not a different build: same everything except the slot count. It exists so +// the configuration join has something to refuse against, since a footprint measured at four slots +// is not a footprint at one -- the count multiplies KV cache. +data one_slot_configuration: ObservedOllamaLaunchConfiguration = observed_ollama_launch_configuration( + server_default_context: token_count(count: 1048576), + serving_slots: positive_slot_count(count: positive_measure_count(predecessor: 0)), +) -data b_iq2_xxs: FixtureBuild = FixtureBuild { release_id: deepseek_v4_flash_release, quant: "IQ2_XXS" as NonEmptyStr, artifact: digest(hex: "32af248f4cab44ffcb631298f66ff0bdf92a9350a3763a86b9a5b304fc4c1e64" as Sha256DigestHex), runtime: ollama_runtime, argv: serving_argv } -data b_iq3_s: FixtureBuild = FixtureBuild { release_id: deepseek_v4_flash_release, quant: "UD-IQ3_S" as NonEmptyStr, artifact: digest(hex: "5b7e1042cc9d3fa6081b4e2d7c3950ae61f8d24b0937ce5182a4f6b0d3e79c18" as Sha256DigestHex), runtime: ollama_runtime, argv: serving_argv } -data b_unmeasured: FixtureBuild = FixtureBuild { release_id: deepseek_v4_flash_release, quant: "UD-IQ4_XS" as NonEmptyStr, artifact: digest(hex: "9d2c48b6103fae57c81d0b2e64937fa5081c3d6e9b204f7a15c8e0b3d629f47a" as Sha256DigestHex), runtime: ollama_runtime, argv: serving_argv } -data b_qwen: FixtureBuild = FixtureBuild { release_id: qwen_release, quant: "Q8_0" as NonEmptyStr, artifact: digest(hex: "a41f70c39b2d85e6014c7f3a92b60d58e7139fac402b6d81c53e97a0f42b18d6" as Sha256DigestHex), runtime: ollama_runtime, argv: serving_argv } +data fixture_configuration: ObservedOllamaLaunchConfiguration = observed_ollama_launch_configuration( + server_default_context: token_count(count: 1048576), + serving_slots: positive_slot_count(count: positive_measure_count(predecessor: 3)), +) + +data b_iq2_xxs: FixtureBuild = FixtureBuild { release_id: deepseek_v4_flash_release, quant: "IQ2_XXS" as NonEmptyStr, artifact: digest(hex: "32af248f4cab44ffcb631298f66ff0bdf92a9350a3763a86b9a5b304fc4c1e64" as Sha256DigestHex), runtime: ollama_runtime, configuration: serving_configuration } +data b_iq3_s: FixtureBuild = FixtureBuild { release_id: deepseek_v4_flash_release, quant: "UD-IQ3_S" as NonEmptyStr, artifact: digest(hex: "5b7e1042cc9d3fa6081b4e2d7c3950ae61f8d24b0937ce5182a4f6b0d3e79c18" as Sha256DigestHex), runtime: ollama_runtime, configuration: serving_configuration } +data b_unmeasured: FixtureBuild = FixtureBuild { release_id: deepseek_v4_flash_release, quant: "UD-IQ4_XS" as NonEmptyStr, artifact: digest(hex: "9d2c48b6103fae57c81d0b2e64937fa5081c3d6e9b204f7a15c8e0b3d629f47a" as Sha256DigestHex), runtime: ollama_runtime, configuration: serving_configuration } +data b_qwen: FixtureBuild = FixtureBuild { release_id: qwen_release, quant: "Q8_0" as NonEmptyStr, artifact: digest(hex: "a41f70c39b2d85e6014c7f3a92b60d58e7139fac402b6d81c53e97a0f42b18d6" as Sha256DigestHex), runtime: ollama_runtime, configuration: serving_configuration } fn fixture_build(quant: NonEmptyStr) -> FixtureBuild { FixtureBuild { release_id: fixture_release, quant: quant, artifact: digest(hex: "ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00ff00" as Sha256DigestHex), runtime: serving_runtime_identity_for_release(rel: ollama_v0_32_9_jetpack6_release), - argv: fixture_argv, + configuration: fixture_configuration, } } @@ -273,17 +313,17 @@ data build_iq2_xxs: QuantizedCandidate = QuantizedCandidate { quality_rank: 2, declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_iq2_xxs, depth: 400060), - fresh_prefill_observations: [ + fresh_prefill_observations: present_prefill(xs: [ prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 160060, rate: 253), prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 255061, rate: 197), prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135), - ], - runtime_memory_observations: [ + ]), + runtime_memory_observations: present_memory(xs: [ iq2_xxs_footprint(depth: 131072, resident: 86532465622), iq2_xxs_footprint(depth: 262144, resident: 87064355798), iq2_xxs_footprint(depth: 400000, resident: 87692389907), iq2_xxs_footprint(depth: 1048576, resident: 88865253620), - ], + ]), } data build_iq3_s: QuantizedCandidate = QuantizedCandidate { @@ -292,10 +332,10 @@ data build_iq3_s: QuantizedCandidate = QuantizedCandidate { declared_context: token_count(count: 1048576), semantic_context: none, fresh_prefill_observations: [], - runtime_memory_observations: [ + runtime_memory_observations: present_memory(xs: [ served(b: b_iq3_s, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 116970000000), - ], + ]), } fn installed() -> List { @@ -309,6 +349,7 @@ data floor_400k: ServingConstraints = ServingConstraints { hot_sessions: 1, prefill_floor: tokens_per_second(count: 0), prefill_regime: FreshSessionPrefill, + configuration: serving_configuration, } // The same floor at four concurrent sessions, so one candidate can be asked the same memory @@ -318,6 +359,7 @@ data floor_400k_four_sessions: ServingConstraints = ServingConstraints { hot_sessions: 4, prefill_floor: tokens_per_second(count: 0), prefill_regime: FreshSessionPrefill, + configuration: serving_configuration, } data floor_8k: ServingConstraints = ServingConstraints { @@ -325,6 +367,7 @@ data floor_8k: ServingConstraints = ServingConstraints { hot_sessions: 1, prefill_floor: tokens_per_second(count: 0), prefill_regime: FreshSessionPrefill, + configuration: serving_configuration, } fn chosen_label(choice: ServingChoice) -> String { @@ -413,6 +456,7 @@ test fn w_the_two_bit_build_is_resident_established_at_the_floor() -> Bool { floor: token_count(count: 400000), hot_sessions: 1, node: serving_node, + configuration: serving_configuration, ) { MemoryFitEstablished { by: o } => token_count_value(t: o.context_depth) == 400000 @@ -437,10 +481,10 @@ test fn w_a_shallow_or_more_concurrent_demand_has_no_qualifying_footprint() -> B let obs = build_iq2_xxs.runtime_memory_observations is_unobserved(evidence: memory_fit_evidence( observations: obs, realization: r_iq2_xxs, - floor: token_count(count: 2000000), hot_sessions: 1, node: serving_node)) + floor: token_count(count: 2000000), hot_sessions: 1, node: serving_node, configuration: serving_configuration)) && is_unobserved(evidence: memory_fit_evidence( observations: obs, realization: r_iq2_xxs, - floor: token_count(count: 400000), hot_sessions: 2, node: serving_node)) + floor: token_count(count: 400000), hot_sessions: 2, node: serving_node, configuration: serving_configuration)) } fn is_unobserved(evidence: MemoryFitEvidence) -> Bool { @@ -468,6 +512,7 @@ fn iq2_xxs_answers_for(r: ServingRealizationIdentity, node: NonEmptyStr) -> Bool floor: token_count(count: 400000), hot_sessions: 1, node: node, + configuration: serving_configuration, ) { MemoryFitEstablished { by: _ } => true MemoryFitRefusedForMemory { by: _ } => false @@ -481,29 +526,22 @@ fn iq2_xxs_answers_for(r: ServingRealizationIdentity, node: NonEmptyStr) -> Bool test fn w_evidence_from_another_realization_or_node_cannot_qualify() -> Bool { let wrong_release = realization( release_id: qwen_release, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_digest, - runtime: r_iq2_xxs.runtime, fixed_mode: r_iq2_xxs.fixed_mode) + runtime: r_iq2_xxs.runtime) let wrong_quant = realization( release_id: r_iq2_xxs.release_id, quant: "IQ3_S" as NonEmptyStr, artifact: r_iq2_xxs.artifact_digest, - runtime: r_iq2_xxs.runtime, fixed_mode: r_iq2_xxs.fixed_mode) + runtime: r_iq2_xxs.runtime) let wrong_artifact = realization( release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, artifact: digest(hex: "7e3b91d0a5c82f461937be04d2a86c50f18e2b7d940c36a1e582f4b0c7d91836" as Sha256DigestHex), - runtime: r_iq2_xxs.runtime, fixed_mode: r_iq2_xxs.fixed_mode) + runtime: r_iq2_xxs.runtime) let wrong_runtime = realization( release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_digest, - runtime: serving_runtime_identity_for_release(rel: ollama_v0_32_9_jetpack5_release), - fixed_mode: r_iq2_xxs.fixed_mode) - let wrong_mode = realization( - release_id: r_iq2_xxs.release_id, quant: r_iq2_xxs.quant_label, artifact: r_iq2_xxs.artifact_digest, - runtime: r_iq2_xxs.runtime, fixed_mode: fixed_runtime_mode_from_argv( - argv: ["ollama" as NonEmptyStr, "serve" as NonEmptyStr, "--flash-attn" as NonEmptyStr], - varied_flags: varied_flags)) + runtime: serving_runtime_identity_for_release(rel: ollama_v0_32_9_jetpack5_release)) iq2_xxs_answers_for(r: r_iq2_xxs, node: serving_node) && !iq2_xxs_answers_for(r: wrong_release, node: serving_node) && !iq2_xxs_answers_for(r: wrong_quant, node: serving_node) && !iq2_xxs_answers_for(r: wrong_artifact, node: serving_node) && !iq2_xxs_answers_for(r: wrong_runtime, node: serving_node) - && !iq2_xxs_answers_for(r: wrong_mode, node: serving_node) && !iq2_xxs_answers_for(r: r_iq2_xxs, node: fixture_node) } @@ -536,6 +574,7 @@ test fn w_a_prefill_receipt_from_another_realization_does_not_qualify() -> Bool hot_sessions: 1, floor: token_count(count: 400000), regime: FreshSessionPrefill, + configuration: serving_configuration, ) { Present { value: _ } => false Absent => true @@ -555,7 +594,7 @@ test fn w_a_prefill_receipt_from_another_realization_does_not_qualify() -> Bool // one, and the ordering claims they support are about the SELECTOR and not about any hardware. data fixture_instrument: NonEmptyStr = "declared fixture, not a reading" as NonEmptyStr -fn fixture_footprint(b: FixtureBuild, resident: Nat) -> OllamaRunnerMemoryObservation { +fn fixture_footprint(b: FixtureBuild, resident: Nat) -> OllamaRunnerMemoryObservation? { served(b: b, node: fixture_node, instrument: fixture_instrument, depth: 1048576, sessions: 1, resident: resident) } @@ -567,7 +606,7 @@ fn fixture_attempt( b: FixtureBuild, depth: Nat, outcome: RunnerAttemptOutcome, -) -> OllamaRunnerMemoryObservation { +) -> OllamaRunnerMemoryObservation? { attempt_of(b: b, node: fixture_node, instrument: fixture_instrument, depth: depth, sessions: 1, outcome: outcome) } @@ -580,10 +619,10 @@ data fixture_low_rank: QuantizedCandidate = QuantizedCandidate { quality_rank: 1, declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_fixture_low, depth: 400060), - fresh_prefill_observations: [ + fresh_prefill_observations: present_prefill(xs: [ prefill_of(b: b_fixture_low, node: fixture_node, sessions: 1, depth: 400060, rate: 500), - ], - runtime_memory_observations: [fixture_footprint(b: b_fixture_low, resident: 28000000000)], + ]), + runtime_memory_observations: present_memory(xs: [fixture_footprint(b: b_fixture_low, resident: 28000000000)]), } data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { @@ -591,10 +630,10 @@ data fixture_high_rank: QuantizedCandidate = QuantizedCandidate { quality_rank: 9, declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_fixture_high, depth: 400060), - fresh_prefill_observations: [ + fresh_prefill_observations: present_prefill(xs: [ prefill_of(b: b_fixture_high, node: fixture_node, sessions: 1, depth: 400060, rate: 400), - ], - runtime_memory_observations: [fixture_footprint(b: b_fixture_high, resident: 38000000000)], + ]), + runtime_memory_observations: present_memory(xs: [fixture_footprint(b: b_fixture_high, resident: 38000000000)]), } test fn w_the_higher_quality_candidate_wins_when_both_are_admissible() -> Bool { @@ -631,12 +670,13 @@ test fn w_a_declared_context_never_qualifies_without_a_retrieval_receipt() -> Bo // (253 -> 197 -> 135 measured on one realization), so a shallow reading flatters the candidate. test fn w_a_shallow_prefill_observation_does_not_answer_a_deeper_floor() -> Bool { match prefill_rate_at_floor( - observations: [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 160060, rate: 253)], + observations: present_prefill(xs: [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 160060, rate: 253)]), realization: r_iq2_xxs, node: serving_node, hot_sessions: 1, floor: token_count(count: 400000), regime: FreshSessionPrefill, + configuration: serving_configuration, ) { Absent => true Present { value: _ } => false @@ -713,19 +753,17 @@ test fn w_nothing_admissible_refuses_and_reports_every_rejection() -> Bool { // regime field at all, so a warm rate is not merely unmatched here, it is unwritable anywhere. The // non-fresh arms of prefill_rate_at_floor have nothing to read. test fn w_a_fresh_prefill_receipt_does_not_answer_a_warm_continuation_floor() -> Bool { - let deep_fresh = [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135)] + let deep_fresh = present_prefill(xs: [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135)]) match prefill_rate_at_floor( observations: deep_fresh, realization: r_iq2_xxs, node: serving_node, hot_sessions: 1, - floor: token_count(count: 400000), regime: WarmContinuation, - ) { + floor: token_count(count: 400000), regime: WarmContinuation, configuration: serving_configuration) { Present { value: _ } => false Absent => match prefill_rate_at_floor( observations: deep_fresh, realization: r_iq2_xxs, node: serving_node, hot_sessions: 1, - floor: token_count(count: 400000), regime: FreshSessionPrefill, - ) { + floor: token_count(count: 400000), regime: FreshSessionPrefill, configuration: serving_configuration) { Absent => false Present { value: r } => tokens_per_second_count(r: r) == 135 } @@ -740,6 +778,7 @@ test fn w_a_warm_floor_over_fresh_only_receipts_is_unanswerable() -> Bool { hot_sessions: 1, prefill_floor: tokens_per_second(count: 1), prefill_regime: WarmContinuation, + configuration: serving_configuration, } match evaluate_candidate(candidate: build_iq2_xxs, node: serving_node, constraints: warm_400k) { ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => @@ -788,12 +827,12 @@ data out_of_memory: RunnerAttemptOutcome = RunnerRefusedForMemory { fn oom_low() -> QuantizedCandidate { with_memory_observations(base: fixture_low_rank, - observations: [fixture_attempt(b: b_fixture_low, depth: 400000, outcome: out_of_memory)]) + observations: present_memory(xs: [fixture_attempt(b: b_fixture_low, depth: 400000, outcome: out_of_memory)])) } fn oom_high() -> QuantizedCandidate { with_memory_observations(base: fixture_high_rank, - observations: [fixture_attempt(b: b_fixture_high, depth: 400000, outcome: out_of_memory)]) + observations: present_memory(xs: [fixture_attempt(b: b_fixture_high, depth: 400000, outcome: out_of_memory)])) } // ================= WHAT AN UNSUCCESSFUL ATTEMPT MEANS ================= @@ -803,11 +842,11 @@ fn oom_high() -> QuantizedCandidate { // whether the weights and cache fit. Only the typed insufficient-memory refusal may reject; the // other failure is UNANSWERABLE and names its cause. test fn w_only_a_typed_memory_refusal_rejects_on_memory() -> Bool { - let broken = with_memory_observations(base: fixture_low_rank, observations: [ + let broken = with_memory_observations(base: fixture_low_rank, observations: present_memory(xs: [ fixture_attempt(b: b_fixture_low, depth: 400000, outcome: RunnerFailedForOtherCause { cause: "chat template rendering failed" as NonEmptyStr, }), - ]) + ])) let rejects_on_memory = match evaluate_candidate( candidate: oom_low(), node: fixture_node, constraints: floor_400k) { ServingCandidateRejected { identity: _, quant_label: _, axis: a } => @@ -871,7 +910,7 @@ fn fixture_served_at( depth: Nat, sessions: Nat, resident: Nat, -) -> OllamaRunnerMemoryObservation { +) -> OllamaRunnerMemoryObservation? { served(b: b, node: fixture_node, instrument: fixture_instrument, depth: depth, sessions: sessions, resident: resident) } @@ -883,10 +922,10 @@ fn fixture_served_at( // would make every complete sweep unanswerable at exactly the floor it measured. test fn w_a_success_at_the_floor_survives_an_oom_above_it() -> Bool { is_admissible(verdict: evaluate_candidate( - candidate: with_memory_observations(base: fixture_low_rank, observations: [ + candidate: with_memory_observations(base: fixture_low_rank, observations: present_memory(xs: [ fixture_served_at(b: b_fixture_low, depth: 400000, sessions: 1, resident: 28000000000), fixture_attempt(b: b_fixture_low, depth: 1048576, outcome: out_of_memory), - ]), + ])), node: fixture_node, constraints: floor_400k)) } @@ -897,10 +936,10 @@ test fn w_a_success_at_the_floor_survives_an_oom_above_it() -> Bool { // quieter -- and the refusal is exactly on point, so it is a memory rejection. A selector that // ignored the concurrency axis on either filter would have to give both demands the same verdict. test fn w_one_candidate_answers_admissible_quiet_and_rejected_busy() -> Bool { - let mixed = with_memory_observations(base: fixture_low_rank, observations: [ + let mixed = with_memory_observations(base: fixture_low_rank, observations: present_memory(xs: [ fixture_served_at(b: b_fixture_low, depth: 400000, sessions: 1, resident: 28000000000), attempt_of(b: b_fixture_low, node: fixture_node, instrument: fixture_instrument, depth: 400000, sessions: 4, outcome: out_of_memory), - ]) + ])) is_admissible(verdict: evaluate_candidate( candidate: mixed, node: fixture_node, constraints: floor_400k)) && is_memory_rejection(verdict: evaluate_candidate( @@ -912,11 +951,11 @@ test fn w_contradictory_attempts_refuse_in_both_orders_and_at_both_magnitudes() let small_failure = fixture_attempt(b: b_fixture_low, depth: 400000, outcome: out_of_memory) is_contradiction_refusal(verdict: evaluate_candidate( candidate: with_memory_observations(base: fixture_low_rank, - observations: [big_success, small_failure]), + observations: present_memory(xs: [big_success, small_failure])), node: fixture_node, constraints: floor_400k)) && is_contradiction_refusal(verdict: evaluate_candidate( candidate: with_memory_observations(base: fixture_low_rank, - observations: [small_failure, big_success]), + observations: present_memory(xs: [small_failure, big_success])), node: fixture_node, constraints: floor_400k)) } @@ -925,7 +964,7 @@ test fn w_contradictory_attempts_refuse_in_both_orders_and_at_both_magnitudes() test fn w_the_large_buffer_success_alone_still_admits() -> Bool { match evaluate_candidate( candidate: with_memory_observations(base: fixture_low_rank, - observations: [fixture_footprint(b: b_fixture_low, resident: 99000000000)]), + observations: present_memory(xs: [fixture_footprint(b: b_fixture_low, resident: 99000000000)])), node: fixture_node, constraints: floor_400k) { ServingCandidateAdmissible { candidate: _, fit: _ } => true ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => false @@ -948,11 +987,11 @@ test fn w_the_large_buffer_success_alone_still_admits() -> Bool { // more of the same machine. This is the direction that carries, and it must still carry. test fn w_a_refusal_generalizes_to_no_easier_point_but_a_success_does() -> Bool { let refused_deep = with_memory_observations(base: fixture_low_rank, - observations: [fixture_attempt(b: b_fixture_low, depth: 1048576, outcome: out_of_memory)]) + observations: present_memory(xs: [fixture_attempt(b: b_fixture_low, depth: 1048576, outcome: out_of_memory)])) let refused_at_point = with_memory_observations(base: fixture_low_rank, - observations: [fixture_attempt(b: b_fixture_low, depth: 400000, outcome: out_of_memory)]) + observations: present_memory(xs: [fixture_attempt(b: b_fixture_low, depth: 400000, outcome: out_of_memory)])) let served_deep = with_memory_observations(base: fixture_low_rank, - observations: [fixture_footprint(b: b_fixture_low, resident: 28000000000)]) + observations: present_memory(xs: [fixture_footprint(b: b_fixture_low, resident: 28000000000)])) is_unobserved_verdict(verdict: evaluate_candidate( candidate: refused_deep, node: fixture_node, constraints: floor_400k)) && is_memory_rejection(verdict: evaluate_candidate( @@ -964,13 +1003,13 @@ test fn w_a_refusal_generalizes_to_no_easier_point_but_a_success_does() -> Bool // AND THE SAME ASYMMETRY ON THE CONCURRENCY AXIS. A refusal at four concurrent sessions says // nothing about one; a success at four establishes one. test fn w_the_concurrency_axis_carries_the_same_asymmetry() -> Bool { - let refused_at_four = with_memory_observations(base: fixture_low_rank, observations: [ + let refused_at_four = with_memory_observations(base: fixture_low_rank, observations: present_memory(xs: [ attempt_of(b: b_fixture_low, node: fixture_node, instrument: fixture_instrument, depth: 400000, sessions: 4, outcome: out_of_memory), - ]) - let served_at_four = with_memory_observations(base: fixture_low_rank, observations: [ + ])) + let served_at_four = with_memory_observations(base: fixture_low_rank, observations: present_memory(xs: [ served(b: b_fixture_low, node: fixture_node, instrument: fixture_instrument, depth: 400000, sessions: 4, resident: 28000000000), - ]) + ])) is_unobserved_verdict(verdict: evaluate_candidate( candidate: refused_at_four, node: fixture_node, constraints: floor_400k)) && is_admissible(verdict: evaluate_candidate( @@ -1021,20 +1060,18 @@ fn is_admissible(verdict: CandidateVerdict) -> Bool { // unmeasured, and if `rate` turns out to be an aggregate rather than per-request it points the // other way. So it must now come back unanswerable. Flip the binding back to `>=` in // slowest_fresh_rate_at_floor and this witness goes red. -data busy_only_prefill: List = [ - prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 4, depth: 400060, rate: 90), -] +data busy_only_prefill: List = present_prefill(xs: [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 4, depth: 400060, rate: 90)]) test fn w_a_busier_prefill_reading_does_not_answer_a_quieter_demand() -> Bool { match prefill_rate_at_floor( observations: busy_only_prefill, realization: r_iq2_xxs, node: serving_node, hot_sessions: 1, - floor: token_count(count: 400000), regime: FreshSessionPrefill) { + floor: token_count(count: 400000), regime: FreshSessionPrefill, configuration: serving_configuration) { Present { value: _ } => false Absent => match prefill_rate_at_floor( observations: busy_only_prefill, realization: r_iq2_xxs, node: serving_node, hot_sessions: 4, - floor: token_count(count: 400000), regime: FreshSessionPrefill) { + floor: token_count(count: 400000), regime: FreshSessionPrefill, configuration: serving_configuration) { Absent => false Present { value: r } => tokens_per_second_count(r: r) == 90 } @@ -1058,13 +1095,12 @@ test fn w_a_prefill_reading_at_lower_concurrency_or_another_node_does_not_qualif fn quiet_rate_at(node: NonEmptyStr, hot: Nat) -> TokensPerSecond? { prefill_rate_at_floor( - observations: [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135)], + observations: present_prefill(xs: [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135)]), realization: r_iq2_xxs, node: node, hot_sessions: hot, floor: token_count(count: 400000), - regime: FreshSessionPrefill, - ) + regime: FreshSessionPrefill, configuration: serving_configuration) } // ================= AN UNANSWERABLE COMPARISON IS NOT A DIFFERENT REALIZATION ================= @@ -1092,7 +1128,7 @@ data b_iq2_xxs_structural_artifact: FixtureBuild = FixtureBuild { quant: b_iq2_xxs.quant, artifact: structural_digest(hex: "32af248f4cab44ff" as Fnv1a64StructuralDigestHex), runtime: b_iq2_xxs.runtime, - argv: b_iq2_xxs.argv, + configuration: b_iq2_xxs.configuration, } fn r_iq2_xxs_structural_artifact() -> ServingRealizationIdentity { @@ -1112,12 +1148,12 @@ fn r_iq2_xxs_structural_artifact() -> ServingRealizationIdentity { // can. The candidate must be ADMISSIBLE on the good reading; the irrelevant one stays irrelevant. // Revert first_incomparable_axis to a node-only filter and this goes red. test fn w_an_irrelevant_cross_family_reading_does_not_stall_the_candidate() -> Bool { - let mixed = with_memory_observations(base: build_iq2_xxs, observations: [ + let mixed = with_memory_observations(base: build_iq2_xxs, observations: present_memory(xs: [ served(b: b_iq2_xxs_structural_artifact, node: serving_node, instrument: ps_instrument, depth: 8192, sessions: 1, resident: 80000000000), served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907), - ]) + ])) is_admissible(verdict: evaluate_candidate( candidate: mixed, node: serving_node, constraints: floor_400k)) } @@ -1159,8 +1195,8 @@ test fn w_a_cross_family_semantic_receipt_refuses_rather_than_reading_as_unverif quality_rank: 1, declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_iq2_xxs_structural_artifact(), depth: 400060), - fresh_prefill_observations: [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135)], - runtime_memory_observations: [served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907)], + fresh_prefill_observations: present_prefill(xs: [prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135)]), + runtime_memory_observations: present_memory(xs: [served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907)]), } refuses_on_artifact_digest(verdict: evaluate_candidate( candidate: c, node: serving_node, constraints: floor_400k)) @@ -1174,8 +1210,8 @@ test fn w_a_cross_family_prefill_receipt_refuses_rather_than_reading_as_unmeasur quality_rank: 1, declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_iq2_xxs, depth: 400060), - fresh_prefill_observations: [prefill_of(b: b_iq2_xxs_structural_artifact, node: serving_node, sessions: 1, depth: 400060, rate: 135)], - runtime_memory_observations: [served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907)], + fresh_prefill_observations: present_prefill(xs: [prefill_of(b: b_iq2_xxs_structural_artifact, node: serving_node, sessions: 1, depth: 400060, rate: 135)]), + runtime_memory_observations: present_memory(xs: [served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907)]), } refuses_on_artifact_digest(verdict: evaluate_candidate( candidate: c, node: serving_node, constraints: floor_400k)) @@ -1188,12 +1224,12 @@ test fn w_a_cross_family_prefill_receipt_refuses_rather_than_reading_as_unmeasur // admissible. Under the point-relevance-only rule it returned MemoryFitIdentityIncomparable, which // made a candidate LESS answerable for holding an extra success. test fn w_a_redundant_incomparable_success_does_not_overturn_established_fit() -> Bool { - let c = with_memory_observations(base: build_iq2_xxs, observations: [ + let c = with_memory_observations(base: build_iq2_xxs, observations: present_memory(xs: [ served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907), served(b: b_iq2_xxs_structural_artifact, node: serving_node, instrument: ps_instrument, depth: 1048576, sessions: 1, resident: 88865253620), - ]) + ])) is_admissible(verdict: evaluate_candidate( candidate: c, node: serving_node, constraints: floor_400k)) } @@ -1202,12 +1238,12 @@ test fn w_a_redundant_incomparable_success_does_not_overturn_established_fit() - // EXACT REFUSAL beside a comparable success is still load-bearing: resolving its identity is exactly // what separates a contradiction from ordinary fit, so it must still stall. test fn w_an_incomparable_exact_refusal_beside_a_success_still_stalls() -> Bool { - let c = with_memory_observations(base: build_iq2_xxs, observations: [ + let c = with_memory_observations(base: build_iq2_xxs, observations: present_memory(xs: [ served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907), attempt_of(b: b_iq2_xxs_structural_artifact, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, outcome: out_of_memory), - ]) + ])) refuses_on_artifact_digest(verdict: evaluate_candidate( candidate: c, node: serving_node, constraints: floor_400k)) } @@ -1242,21 +1278,115 @@ test fn w_two_failures_differing_only_in_cause_report_the_same_one_in_both_order let transport = attempt_of(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, outcome: RunnerFailedForOtherCause { cause: "transport reset" as NonEmptyStr }) - match failure_cause_reported(observations: [template, transport]) { + match failure_cause_reported(observations: present_memory(xs: [template, transport])) { Absent => false Present { value: a } => - match failure_cause_reported(observations: [transport, template]) { + match failure_cause_reported(observations: present_memory(xs: [transport, template])) { Absent => false Present { value: b } => (a as String) == (b as String) } } } +// ============ THE TWO FALSIFIERS THE SIDE CHAT NAMED, AS DISCRIMINATING REDS ============ +// +// A stall rule that only fires when the incomparable row would be SELECTED as the carrier is not a +// stall rule -- it answers from the population it happens to find comparable. Both fixtures below +// hold a comparable reading that alone answers the question, beside an incomparable reading that +// would change the answer if it belonged to this candidate. The identity question is therefore +// load-bearing in both, and the only honest verdict is that it cannot be resolved. + +// The comparable reading clears the floor (135 >= 100); the incomparable one does not (50 < 100). +// Answering 135 asserts the 50 belongs to somebody else -- which is exactly what could not be +// decided. +data floor_400k_prefill_100: ServingConstraints = ServingConstraints { + context_floor: token_count(count: 400000), + hot_sessions: 1, + prefill_floor: tokens_per_second(count: 100), + prefill_regime: FreshSessionPrefill, + configuration: serving_configuration, +} + +test fn w_a_slower_incomparable_prefill_beside_a_passing_one_stalls_at_the_rate_floor() -> Bool { + let rows = [ + prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135), + prefill_of(b: b_iq2_xxs_structural_artifact, node: serving_node, sessions: 1, depth: 400060, rate: 50), + ] + let c = QuantizedCandidate { + realization: r_iq2_xxs, + quality_rank: 1, + declared_context: token_count(count: 1048576), + semantic_context: verified_at(r: r_iq2_xxs, depth: 400060), + fresh_prefill_observations: present_prefill(xs: rows), + runtime_memory_observations: present_memory(xs: [served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907)]), + } + refuses_on_artifact_digest(verdict: evaluate_candidate( + candidate: c, node: serving_node, constraints: floor_400k_prefill_100)) +} + +// Two non-memory failures at the same exact point, one comparable and one not. Reporting a cause at +// all asserts the failure belongs to this candidate; reporting the INCOMPARABLE row's cause asserts +// it about a row whose identity was never established. Either way the identity question comes first, +// and it must come first in both roster orders. +fn incomparable_beside_comparable_failure(reversed: Bool) -> Bool { + let comparable = attempt_of(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, + depth: 400000, sessions: 1, + outcome: RunnerFailedForOtherCause { cause: "z" as NonEmptyStr }) + let unalignable = attempt_of(b: b_iq2_xxs_structural_artifact, node: serving_node, + instrument: ps_instrument, depth: 400000, sessions: 1, + outcome: RunnerFailedForOtherCause { cause: "a" as NonEmptyStr }) + let rows = if reversed { [unalignable, comparable] } else { [comparable, unalignable] } + refuses_on_artifact_digest(verdict: evaluate_candidate( + candidate: with_memory_observations(base: build_iq2_xxs, observations: present_memory(xs: rows)), + node: serving_node, constraints: floor_400k)) +} + +// THE POSITIVE CONTROLS FOR BOTH FALSIFIERS. Make the unalignable row comparable -- same digest, +// nothing else changed -- and each pair answers rather than stalling: the 50 tok/s reading is then +// this candidate's own and loses on the rate floor, and the failure pair reports a cause. So the +// refusals above are the unresolved identity and not the presence of a second row. +test fn w_the_same_two_prefill_rows_answer_once_their_identity_is_comparable() -> Bool { + let c = QuantizedCandidate { + realization: r_iq2_xxs, + quality_rank: 1, + declared_context: token_count(count: 1048576), + semantic_context: verified_at(r: r_iq2_xxs, depth: 400060), + fresh_prefill_observations: present_prefill(xs: [ + prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 135), + prefill_of(b: b_iq2_xxs, node: serving_node, sessions: 1, depth: 400060, rate: 50), + ]), + runtime_memory_observations: present_memory(xs: [served(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907)]), + } + match evaluate_candidate(candidate: c, node: serving_node, constraints: floor_400k_prefill_100) { + ServingCandidateRejected { identity: _, quant_label: _, axis: _ } => true + ServingCandidateAdmissible { candidate: _, fit: _ } => false + ServingCandidateUnanswerable { identity: _, quant_label: _, missing: _ } => false + } +} + +test fn w_the_same_two_failures_report_a_cause_once_their_identity_is_comparable() -> Bool { + let comparable = attempt_of(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, + depth: 400000, sessions: 1, + outcome: RunnerFailedForOtherCause { cause: "z" as NonEmptyStr }) + let alignable = attempt_of(b: b_iq2_xxs, node: serving_node, instrument: ps_instrument, + depth: 400000, sessions: 1, + outcome: RunnerFailedForOtherCause { cause: "a" as NonEmptyStr }) + match failure_cause_reported(observations: present_memory(xs: [comparable, alignable])) { + Present { value: _ } => true + Absent => false + } +} + +test fn w_an_incomparable_exact_failure_beside_a_comparable_one_stalls_in_both_orders() -> Bool { + incomparable_beside_comparable_failure(reversed: false) + && incomparable_beside_comparable_failure(reversed: true) +} + test fn w_a_cross_family_digest_refuses_rather_than_reading_as_unobserved() -> Bool { - let cross_family = with_memory_observations(base: build_iq2_xxs, observations: [ + let cross_family = with_memory_observations(base: build_iq2_xxs, observations: present_memory(xs: [ served(b: b_iq2_xxs_structural_artifact, node: serving_node, instrument: ps_instrument, depth: 400000, sessions: 1, resident: 87692389907), - ]) + ])) match evaluate_candidate(candidate: cross_family, node: serving_node, constraints: floor_400k) { ServingCandidateUnanswerable { identity: _, quant_label: _, missing: m } => match m { @@ -1283,7 +1413,6 @@ test fn w_realization_comparison_reaches_all_three_states_and_different_wins() - quant: r_iq2_xxs.quant_label, artifact: structural_digest(hex: "32af248f4cab44ff" as Fnv1a64StructuralDigestHex), runtime: ollama_runtime, - fixed_mode: serial_mode, ) is_same(c: compare_serving_realization(a: r_iq2_xxs, b: r_iq2_xxs)) && is_different(c: compare_serving_realization(a: r_iq2_xxs, b: r_qwen)) @@ -1291,30 +1420,68 @@ test fn w_realization_comparison_reaches_all_three_states_and_different_wins() - && is_different(c: compare_serving_realization(a: r_iq2_xxs, b: other_release_cross_family)) } +// DERIVED THROUGH THE ELIMINATOR, NOT RE-MATCHED. Each of these hand-matched RealizationComparison, +// which is a second classification of a coproduct that already has one -- so a fourth arm would leave +// them silently answering `false` instead of failing to compile. They now route through +// realization_comparison_fold, the single eliminator in gunbc.model.choice. fn is_same(c: RealizationComparison) -> Bool { - match c { - RealizationSame => true - RealizationDifferent => false - RealizationIncomparable { axis: _ } => false - } + realization_comparison_fold(comparison: c, same: true, different: false, incomparable: false) } fn is_different(c: RealizationComparison) -> Bool { - match c { - RealizationSame => false - RealizationDifferent => true - RealizationIncomparable { axis: _ } => false - } + realization_comparison_fold(comparison: c, same: false, different: true, incomparable: false) } fn is_incomparable(c: RealizationComparison) -> Bool { - match c { - RealizationSame => false - RealizationDifferent => false - RealizationIncomparable { axis: _ } => true + realization_comparison_fold(comparison: c, same: false, different: false, incomparable: true) +} + +// =============== THE POINT IS A RECEIPT, NOT AN AUTHORED PAIR =============== +// +// serving_configuration is the fleet's real unit: 1,048,576 server-default context, 4 slots. A +// receipt is admitted only when the execution it describes is one this unit could have performed. +fn attempt_from_receipt(r: ExecutedRequestReceipt) -> OllamaRunnerMemoryObservation? { + observed_memory_attempt( + release_id: b_iq2_xxs.release_id, quant_label: b_iq2_xxs.quant, + artifact_digest: b_iq2_xxs.artifact, runtime: b_iq2_xxs.runtime, + configuration: serving_configuration, + node: serving_node, instrument: ps_instrument, receipt: r, + outcome: out_of_memory) +} + +fn receipt_admits(r: ExecutedRequestReceipt) -> Bool { + match attempt_from_receipt(r: r) { + Present { value: _ } => true + Absent => false } } +fn plain_receipt(depth: Nat, sessions: Nat) -> ExecutedRequestReceipt { + executed_request_receipt( + prompt_eval_count: depth, concurrent_requests: sessions, instrument: ps_instrument, + context_override: NoModelOrRequestContextOverride) +} + +test fn w_a_point_its_configuration_could_not_have_produced_has_no_observation() -> Bool { + receipt_admits(r: plain_receipt(depth: 400000, sessions: 1)) + && !receipt_admits(r: plain_receipt(depth: 2000000, sessions: 1)) + && !receipt_admits(r: plain_receipt(depth: 400000, sessions: 5)) + && !receipt_admits(r: plain_receipt(depth: 400000, sessions: 0)) +} + +// THE SERVER DEFAULT IS NOT A CEILING, and this is the reading that proves the rename was not +// cosmetic. The same 2,000,000-token receipt that is refused above is ADMITTED when it carries an +// observed request-level override raising the effective context past the server default. A bound +// named "ceiling" has no arm that could do this. +test fn w_an_observed_request_override_moves_the_bound_the_server_default_set() -> Bool { + receipt_admits(r: executed_request_receipt( + prompt_eval_count: 2000000, concurrent_requests: 1, instrument: ps_instrument, + context_override: RequestContextOverrideObserved { effective_context: token_count(count: 2000000) })) + && !receipt_admits(r: executed_request_receipt( + prompt_eval_count: 2000000, concurrent_requests: 1, instrument: ps_instrument, + context_override: RequestContextOverrideObserved { effective_context: token_count(count: 500000) })) +} + test fn w_all_serving_choice_claims_hold() -> Bool { w_at_the_400k_floor_the_higher_quality_build_fits_and_is_still_unresolved() && w_the_real_roster_refuses_while_the_higher_rank_is_unresolved() @@ -1346,6 +1513,12 @@ test fn w_all_serving_choice_claims_hold() -> Bool { && w_an_incomparable_exact_refusal_beside_a_success_still_stalls() && w_two_failures_differing_only_in_cause_report_the_same_one_in_both_orders() && w_an_irrelevant_cross_family_reading_does_not_stall_the_candidate() + && w_a_slower_incomparable_prefill_beside_a_passing_one_stalls_at_the_rate_floor() + && w_an_incomparable_exact_failure_beside_a_comparable_one_stalls_in_both_orders() + && w_the_same_two_prefill_rows_answer_once_their_identity_is_comparable() + && w_the_same_two_failures_report_a_cause_once_their_identity_is_comparable() + && w_a_point_its_configuration_could_not_have_produced_has_no_observation() + && w_an_observed_request_override_moves_the_bound_the_server_default_set() && w_realization_comparison_reaches_all_three_states_and_different_wins() } @@ -1360,9 +1533,9 @@ data build_unmeasured: QuantizedCandidate = QuantizedCandidate { quality_rank: 4, declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_unmeasured, depth: 400060), - fresh_prefill_observations: [ + fresh_prefill_observations: present_prefill(xs: [ prefill_of(b: b_unmeasured, node: serving_node, sessions: 1, depth: 400060, rate: 135), - ], + ]), runtime_memory_observations: [], } @@ -1406,13 +1579,13 @@ data other_release: QuantizedCandidate = QuantizedCandidate { quality_rank: 3, declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_qwen, depth: 400060), - fresh_prefill_observations: [ + fresh_prefill_observations: present_prefill(xs: [ prefill_of(b: b_qwen, node: serving_node, sessions: 1, depth: 400060, rate: 900), - ], - runtime_memory_observations: [ + ]), + runtime_memory_observations: present_memory(xs: [ served(b: b_qwen, node: serving_node, instrument: fixture_instrument, depth: 1048576, sessions: 1, resident: 30600000000), - ], + ]), } test fn w_two_admissible_releases_refuse_for_want_of_a_cross_release_order() -> Bool { @@ -1457,10 +1630,10 @@ data fixture_tied_a: QuantizedCandidate = QuantizedCandidate { quality_rank: 5, declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_tied_a, depth: 400060), - fresh_prefill_observations: [ + fresh_prefill_observations: present_prefill(xs: [ prefill_of(b: b_tied_a, node: fixture_node, sessions: 1, depth: 400060, rate: 500), - ], - runtime_memory_observations: [fixture_footprint(b: b_tied_a, resident: 28000000000)], + ]), + runtime_memory_observations: present_memory(xs: [fixture_footprint(b: b_tied_a, resident: 28000000000)]), } data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { @@ -1468,10 +1641,10 @@ data fixture_tied_b: QuantizedCandidate = QuantizedCandidate { quality_rank: 5, declared_context: token_count(count: 1048576), semantic_context: verified_at(r: r_tied_b, depth: 400060), - fresh_prefill_observations: [ + fresh_prefill_observations: present_prefill(xs: [ prefill_of(b: b_tied_b, node: fixture_node, sessions: 1, depth: 400060, rate: 500), - ], - runtime_memory_observations: [fixture_footprint(b: b_tied_b, resident: 29000000000)], + ]), + runtime_memory_observations: present_memory(xs: [fixture_footprint(b: b_tied_b, resident: 29000000000)]), } fn is_tie_refusal(choice: ServingChoice) -> Bool { From 9a0334483810174343b38a484aedc2576d5a563d Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Wed, 2 Sep 2026 05:08:39 +0000 Subject: [PATCH 22/22] A parameter count carries its scale in the type, not in the reader's head ParameterCounts stored base_total, activated_per_token and attached_speculative as bare Nat, and the fixture that established the module wrote 284 to mean 284 BILLION. A factor of a billion carried entirely by prose, in a module whose whole subject is which quantization fits which host. ParameterCount joins its four siblings in std.measure as Measure. Mega rather than Giga because the published figure is not an integer number of billions -- DeepSeek-V4's own /api/show reports 284.3B, which Giga over Nat cannot write and silently rounds to 284. At Mega it is 284300 and exact. The stage0 std_measure.rs mirror is regenerated by the regen actuator; the other four files that actuator rewrites are left alone, their drift being inherited from main rather than caused here. Raised by review 58449. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01U397y4s3dSBof7vGPAX87G --- dag/gunbc/model/publication.dag | 13 +++++++--- dag/std/measure.dag | 25 ++++++++++++++++++ ...odel_population_narrowing_witness_test.dag | 8 +++--- src/v1/stage0/src/std_measure.rs | 26 +++++++++++++++++++ 4 files changed, 64 insertions(+), 8 deletions(-) diff --git a/dag/gunbc/model/publication.dag b/dag/gunbc/model/publication.dag index 39cbd7969cc..233e5934115 100644 --- a/dag/gunbc/model/publication.dag +++ b/dag/gunbc/model/publication.dag @@ -3,7 +3,7 @@ module gunbc.model.publication import std.types { String, Bool, List, NonEmptyStr, Int, Timestamp } import std.content_hash { ContentHash, serialize_content_hash } import std.nat { Nat } -import std.measure { TokenCount } +import std.measure { TokenCount, ParameterCount } // WHAT A MODEL RELEASE IS, modeled independently of anyone who serves, packages or hosts it. // @@ -54,10 +54,15 @@ type WeightLicense // // activated is the per-token count for a mixture-of-experts release and equals total for a dense // one. It governs decode bandwidth; total governs whether the release is resident at all. +// THE SCALE IS IN THE TYPE, NOT IN THE READER'S HEAD. These were bare `Nat`s, and the fixture that +// established this module wrote `base_total: 284` to mean 284 BILLION -- a factor of a billion +// carried entirely by prose, in a module whose whole subject is which quantization fits which host. +// A ParameterCount is Measure from std.measure, so 284.3B is written as 284300 and +// means one thing. type ParameterCounts { - base_total: Nat - activated_per_token: Nat - attached_speculative: Nat + base_total: ParameterCount + activated_per_token: ParameterCount + attached_speculative: ParameterCount } // WEIGHTS AS A PUBLISHED FACT: a repository, a revision, a license. This is what makes a release diff --git a/dag/std/measure.dag b/dag/std/measure.dag index 65aaeda502a..6113a6e5d62 100644 --- a/dag/std/measure.dag +++ b/dag/std/measure.dag @@ -909,6 +909,31 @@ fn character_count_value(c: CharacterCount) -> Nat { // TokenCount where a CharacterCount belongs is WRITABLE — the family buys reading clarity and one // construction site each, never a wall. +// A count of MODEL PARAMETERS, a fifth member of the same Count family -- and the one member whose +// SCALE is the whole point. A published parameter figure is quoted in billions and is not an +// integer number of them: DeepSeek-V4's own /api/show reports 284.3B. Stored at Giga over Nat that +// fact is unwritable and rounds to 284; stored as a bare Nat the scale lives in prose, which is how +// `284` comes to mean 284,000,000,000 with nothing to say so. Mega holds it exactly -- 284.3B is +// 284300 -- so the fraction survives without introducing a rational. +// +// It lives HERE for the reason its four siblings do: Measure of Count at Mega over Nat takes only +// arguments std.measure already owns, so an instantiation declared downstream would be a fifth name +// for a shape this module already names. Landed by review 58449 on gunbc#9897, which caught +// gunbc.model.publication's ParameterCounts storing scaled quantities as raw naturals. +// +// HONEST LIMIT, the same one the siblings carry: the family is structurally identical, so passing a +// ParameterCount where a CharacterCount belongs is WRITABLE. It buys reading clarity and one +// construction site, never a wall. +type ParameterCount = Measure + +fn parameter_count(millions: Nat) -> ParameterCount { + Measure { count: millions } +} + +fn parameter_count_value(p: ParameterCount) -> Nat { + measure_count(p) +} + fn token_count(count: Nat) -> TokenCount { Measure { count: count } } diff --git a/dag/test/claim/model/model_population_narrowing_witness_test.dag b/dag/test/claim/model/model_population_narrowing_witness_test.dag index db2b460eeff..fa2fc11ef62 100644 --- a/dag/test/claim/model/model_population_narrowing_witness_test.dag +++ b/dag/test/claim/model/model_population_narrowing_witness_test.dag @@ -2,7 +2,7 @@ module test.claim.model.model_population_narrowing_witness_test import std.types { String, Bool, List, NonEmptyStr, Int, Timestamp } import std.nat { Nat } -import std.measure { TokenCount, token_count } +import std.measure { TokenCount, token_count, parameter_count } import gunbc.model.publication { ModelRelease, OpenWeightRelease, ClosedWeightRelease, ReleaseIdentity, WeightPublication, ParameterCounts, DeclaredContext, @@ -62,9 +62,9 @@ data deepseek_v4_flash: ModelRelease = OpenWeightRelease { revision: "main" as NonEmptyStr, license: MitLicense, parameters: ParameterCounts { - base_total: 284, - activated_per_token: 13, - attached_speculative: 20, + base_total: parameter_count(millions: 284300), + activated_per_token: parameter_count(millions: 13000), + attached_speculative: parameter_count(millions: 20000), }, }, declared_context: DeclaredContext { diff --git a/src/v1/stage0/src/std_measure.rs b/src/v1/stage0/src/std_measure.rs index 52376ac791d..5f165fbc261 100644 --- a/src/v1/stage0/src/std_measure.rs +++ b/src/v1/stage0/src/std_measure.rs @@ -953,6 +953,19 @@ pub fn character_count_value(c: CharacterCount) -> Nat { measure_count(c.clone()) } +pub type ParameterCount = Rc>; + +pub fn parameter_count(millions: Nat) -> ParameterCount { + Rc::new(Measure { + count: millions.clone(), + _phantom: std::marker::PhantomData, + }) +} + +pub fn parameter_count_value(p: ParameterCount) -> Nat { + measure_count(p.clone()) +} + pub fn token_count(count: Nat) -> TokenCount { Rc::new(Measure { count: count.clone(), @@ -964,6 +977,19 @@ pub fn token_count_value(t: TokenCount) -> Nat { measure_count(t.clone()) } +pub type TokensPerSecond = Rc>; + +pub fn tokens_per_second(count: Nat) -> TokensPerSecond { + Rc::new(Measure { + count: count.clone(), + _phantom: std::marker::PhantomData, + }) +} + +pub fn tokens_per_second_count(r: TokensPerSecond) -> Nat { + measure_count(r.clone()) +} + pub fn millicore(count: Nat) -> Millicore { Rc::new(Measure { count: count.clone(),