From 05778a38b21acd9f7fc7e5720d4b5bb84e02c647 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 19:08:00 +0000 Subject: [PATCH 1/4] Model the concurrency axis, name the context value a default rather than a limit, and file the drop Three fleet-serving facts that were wrong, missing, or misnamed. MISNAMED. extdeps.ollama.server_env called OLLAMA_CONTEXT_LENGTH "the context window the SERVER allocates" and "the CONFIGURED limit". Upstream 0.32.9 resolves options in layers -- environment, then model, then request -- and an explicit num_ctx at either later layer replaces it. So it never limited anything; it is the value used when no override applies. One name carried two materially different contracts, and a deployment reading it as a ceiling would believe it had bounded something any caller can raise. Renamed through the carriers to default_context. MISSING. OLLAMA_NUM_PARALLEL was not in the closed variable set and not rendered, so the hosts inherited whatever the runtime chose -- one slot, which SERIALIZES. On 2026-09-01 a ~61,000-token prefill held one host for roughly two hundred seconds while every other caller queued, and the queue was read as a latency regression in the model. The axis is now modeled and the desired value is 4. WRONG. The P1b note said the two axes trade "because the slot count divides the context window". Measured on one host, one model, changing only that variable: one slot reported context_length 1048576 and 88,865,253,620 bytes; two slots reported context_length 1048576 and 90,543,761,652. Each slot gets its own full window. The count MULTIPLIES memory; it does not DIVIDE context. Deleted rather than softened. The desired context becomes 1,048,576, and its evidence sentence is replaced. The old one cited an /api/ps reading as proof of a configured value -- but /api/ps reports what a loaded runner holds, and both hosts carried no OLLAMA_CONTEXT_LENGTH line at all, so the number being confirmed was the runtime's own default. A declaration cited its own absence as confirmation. AND THE DROP IS FILED, one row for both values because they are one defect in two spellings: measured on one node, one artifact, one runtime, one concurrency condition, applied to two hosts and every realization the cell loads. The trigger names the capability rather than an artifact -- an observation table alone does not discharge it, because while a global default exists an unmeasured realization still inherits these numbers automatically. An unobserved realization must refuse, not inherit. --- dag/extdeps/ollama/server_env.dag | 56 +++++++-- dag/gunbc/spark/serving_desired.dag | 107 +++++++++++++++--- dag/gunbc/spark/serving_install_paths.dag | 13 ++- dag/gunbc/spark/serving_membership.dag | 6 +- dag/gunbc/spark/serving_release.dag | 3 +- dag/gunbc/spark/serving_unit_render.dag | 13 ++- .../spark_serving_release_witness_test.dag | 10 +- ...spark_serving_unit_render_witness_test.dag | 8 +- 8 files changed, 169 insertions(+), 47 deletions(-) diff --git a/dag/extdeps/ollama/server_env.dag b/dag/extdeps/ollama/server_env.dag index 7da4dd04e70..c90b674d838 100644 --- a/dag/extdeps/ollama/server_env.dag +++ b/dag/extdeps/ollama/server_env.dag @@ -1,6 +1,7 @@ module extdeps.ollama.server_env import std.types { NonEmptyStr, String } +import std.nat { Nat } import std.measure { TokenCount, token_count_value } import extdeps.external_authority { ExternalAuthority, ExternalModelScope, ExternalSubjectRef } import extdeps.uri { Uri, Https } @@ -44,12 +45,14 @@ type OllamaServerEnvVariable = OllamaHost | OllamaModels | OllamaContextLength + | OllamaNumParallel fn ollama_server_env_name(variable: OllamaServerEnvVariable) -> NonEmptyStr { match variable { OllamaHost => "OLLAMA_HOST" as NonEmptyStr OllamaModels => "OLLAMA_MODELS" as NonEmptyStr OllamaContextLength => "OLLAMA_CONTEXT_LENGTH" as NonEmptyStr + OllamaNumParallel => "OLLAMA_NUM_PARALLEL" as NonEmptyStr } } @@ -71,21 +74,50 @@ fn ollama_models_env_assignment(root: NonEmptyStr) -> NonEmptyStr { ollama_server_env_assignment(variable: OllamaModels, value: root) } -// The context window the SERVER allocates for a loaded model, in tokens. +// THE SERVER'S DEFAULT CONTEXT WINDOW, in tokens -- a default, not a limit, and the distinction is +// this function's whole subject. // -// The server decides what is allocated; the model file only declares what it was trained to -// support. So a deployment that wants a model's full declared window has to say so, and one that -// says nothing gets Ollama's own default rather than the model's declaration. Which window a given -// deployment wants is a deployment fact, so it is a parameter here and never a default -- the same -// reason `ollama_host_env_assignment` refuses to spell a bind address. +// WHAT THIS SAID BEFORE, AND WHY IT WAS A MEANING FORK. It read that this is "the context window the +// SERVER allocates" and "the CONFIGURED limit". Upstream 0.32.9 resolves options in layers -- the +// environment-derived value first, then the model's own options, then the request's -- and an +// explicit num_ctx at either later layer REPLACES this one. So it never limited anything: it is the +// value the server uses when no model or request override applies. Calling it a limit gave one name +// two materially different contracts, and a deployment reading it as a ceiling would believe it had +// bounded something a caller can raise at will. // -// This is the CONFIGURED limit, which is not the EFFECTIVE one. What a loaded model actually -// allocated is a separate observation read back from the running server; a request the host cannot -// satisfy is resolved by the runtime, not by this assignment, so the two are different facts and -// must not be collapsed into one. -fn ollama_context_length_env_assignment(configured_context: TokenCount) -> NonEmptyStr { +// Which default a deployment wants is a deployment fact, so it is a parameter here and never a +// default of ours -- the same reason `ollama_host_env_assignment` refuses to spell a bind address. +// +// THE DEFAULT IS NOT THE EFFECTIVE VALUE, and that separation is what a reader most needs. What a +// loaded runner actually holds is `/api/ps.context_length`, a distinct observation with its own +// instrument. This assignment cannot be evidence for that reading, and that reading cannot be +// evidence for this assignment -- a measured 2026-09-01 case had both hosts serving a value this +// assignment had never been rendered onto them at all, and an /api/ps reading was cited in the +// consumer as confirmation of a configured value it could not have discriminated. +fn ollama_context_length_env_assignment(default_context: TokenCount) -> NonEmptyStr { ollama_server_env_assignment( variable: OllamaContextLength, - value: to_string(token_count_value(t: configured_context)) as NonEmptyStr, + value: to_string(token_count_value(t: default_context)) as NonEmptyStr, + ) +} + +// HOW MANY REQUESTS ONE LOADED MODEL SERVES CONCURRENTLY. Unset, Ollama chooses for itself, and a +// host that chose one slot SERIALIZES: a long prefill blocks every other caller for its whole +// duration, which is not a throughput cost but a queue. +// +// THE SLOT COUNT DOES NOT DIVIDE THE CONTEXT WINDOW. That is worth stating in the authority because +// the opposite was written down and believed. Measured on one host, one model, changing only this +// variable: at one slot the runner reported context_length 1048576 and 88,865,253,620 bytes; at two +// slots it reported context_length 1048576 and 90,543,761,652 bytes. Each slot receives its OWN full +// window and costs its own cache. The count multiplies memory; it does not divide context. +// +// So the two axes do trade inside one memory budget, but multiplicatively rather than by division, +// and the per-slot price is a property of the REALIZATION -- a compressed-cache architecture pays +// little per slot and an uncompressed one pays a great deal. A slot count chosen against one model's +// per-slot cost is not evidence for another's. +fn ollama_num_parallel_env_assignment(slots: Nat) -> NonEmptyStr { + ollama_server_env_assignment( + variable: OllamaNumParallel, + value: to_string(slots) as NonEmptyStr, ) } diff --git a/dag/gunbc/spark/serving_desired.dag b/dag/gunbc/spark/serving_desired.dag index 6fb1cf1fa77..2d200e71901 100644 --- a/dag/gunbc/spark/serving_desired.dag +++ b/dag/gunbc/spark/serving_desired.dag @@ -2,6 +2,10 @@ module gunbc.spark.serving_desired import std.types { String, Bool, List, NonEmptyStr } import std.measure { TokenCount, token_count } +import std.nat { Nat } +import gunbc.guarantee_rung_drop { + GuaranteeRungDrop, MechanicallyPreventable, Mitigatable, DeletedWithoutReplacement, +} import product.placement_supply { HostIdentity } import gunbc.spark.cell_role { spark_serving_cell_hosts } import gunbc.ollama_runtime_bundle { ollama_runtime_materialization_identity_for_required_release } @@ -237,32 +241,69 @@ data spark_serving_fleet_desired: SparkServingFleetDesired = spark_serving_fleet fn spark_serving_desired_launch_profile() -> OllamaServingLaunchProfile { OllamaServingLaunchProfile { bind: spark_serving_desired_bind_listen, - configured_context: spark_serving_desired_context_length, + default_context: spark_serving_desired_context_length, + serving_slots: spark_serving_desired_serving_slots_value, } } -// THE CONFIGURED CONTEXT WINDOW, AUTHORED AS THE MIGRATION DECISION. +// THE SERVER'S DEFAULT CONTEXT WINDOW, AUTHORED AS THE MIGRATION DECISION. +// +// TWO VALUES STOOD HERE BEFORE AND NEITHER WAS EVER ENACTED. The 8192 was never rendered into the +// unit at all. The 131072 that replaced it was rendered by this module's own renderer and still did +// not reach either host: read directly on 2026-09-01, the deployed user unit on spark-a3ee and +// spark-3bd5 carried three Environment lines and no OLLAMA_CONTEXT_LENGTH among them. +// +// THE EVIDENCE SENTENCE IS WHY THAT WENT UNNOTICED FOR SO LONG, and it is the more useful half of +// this history. It read that the wet observation was EVIDENCE and not source, citing both hosts +// reporting `context_length: 131072` from /api/ps. But /api/ps reports what a LOADED RUNNER holds, +// which extdeps.ollama.server_env states is a different fact from the server default and must not be +// collapsed into it. An effective reading was admitted as evidence for a configured value, and it +// could not have discriminated: with no assignment on either host the number being confirmed was the +// runtime's own automatic default. A declaration cited its own absence as confirmation. +// +// 1048576 IS SELECTED, and what makes it defensible is a load rather than a reading: the served +// build declares `deepseek4.context_length` 1048576 and at that full window its runner reports +// 88,865,253,620 bytes of buffer. Re-derive by loading at an explicit num_ctx and reading /api/ps +// back; that instrument, not this number, is the authority for what any build costs. +// +// A smaller default remains writable. A larger one than a model declares is a real request the +// runtime is free to refuse -- Ollama clamps to the model's declared maximum, so this is a CEILING +// ACROSS THE ROSTER and never a per-model promise. +data spark_serving_desired_context_length: TokenCount = token_count(count: 1048576) + +// HOW MANY REQUESTS ONE LOADED MODEL SERVES CONCURRENTLY -- the P1b row, now authored. // -// The 8192 that stood here was never enacted: nothing rendered the capacity policy into the unit, -// so the declaration sat beside a process it did not configure. Making the carrier causal without -// reselecting the value would therefore not have preserved behavior -- it would have SHRUNK the -// live window 16x on the first converge, as a side effect of a repair whose subject was causality -// and not capacity. +// THE NOTE ABOVE SAID THIS TRADED AGAINST CONTEXT "BECAUSE THE SLOT COUNT DIVIDES THE CONTEXT +// WINDOW". That sentence is deleted rather than softened, because it is false. Measured on one host, +// one model, changing only OLLAMA_NUM_PARALLEL: at one slot the runner reported context_length +// 1048576 and 88,865,253,620 bytes of buffer; at two slots it reported context_length 1048576 and +// 90,543,761,652. Each slot receives its own full window. THE COUNT MULTIPLIES MEMORY; IT DOES NOT +// DIVIDE CONTEXT. The two axes do trade inside one budget, but multiplicatively. // -// 131072 is SELECTED here, and the wet observation is its EVIDENCE, not its source: both hosts were -// read reporting `context_length: 131072` from the Ollama /api/ps surface, which establishes that -// this value preserves what the running service already exposes. Observed state does not silently -// become desired state -- an operator chose this number and the reading is why it is defensible. A -// smaller window remains writable, and a larger one than the model declares is a real request the -// runtime is free to refuse. -data spark_serving_desired_context_length: TokenCount = token_count(count: 131072) +// WHY THE VALUE IS NOT 1. Unset, the runtime chose a single slot, and a single slot SERIALIZES: on +// 2026-09-01 a ~61,000-token prefill occupied one host for roughly two hundred seconds while every +// other caller queued, and the queue was read as a latency regression in the model rather than as +// contention. That is not a throughput cost, it is a head-of-line block, and the harness workload +// this cell exists to serve is concurrent by construction. +// +// WHAT THIS VALUE IS NOT EVIDENCE FOR, and the reason a rung drop is filed below rather than a +// footnote written here: the per-slot price is a property of the REALIZATION. The measured +// 1,678,508,032 bytes is a compressed-cache figure for one DeepSeek build; an architecture without +// that compression pays far more per slot, and llama4:scout in particular declares a window far +// beyond this ceiling with no such compression. A count chosen against one model's per-slot cost is +// not evidence for another's, and this row applies to every realization the cell loads. +data spark_serving_desired_serving_slots_value: Nat = 4 // The unit's window is DERIVED from the serving policy rather than authored beside it. This is the // causal link the repair adds: the policy is the single authority, the unit text is a function of // it, and the unit's content identity is a function of the text -- so a policy change cannot fail // to reach the file, and cannot reach it without producing a member delta. -fn spark_serving_desired_configured_context() -> TokenCount { - spark_serving_desired_launch_profile().configured_context +fn spark_serving_desired_serving_slots() -> Nat { + spark_serving_desired_launch_profile().serving_slots +} + +fn spark_serving_desired_default_context() -> TokenCount { + spark_serving_desired_launch_profile().default_context } fn spark_serving_terminal_intent() -> SparkServingIntent { @@ -283,3 +324,37 @@ fn spark_serving_automation_principal_login() -> NonEmptyStr { fn spark_serving_automation_principal() -> NamedPrincipal { NamedPrincipal { login: spark_serving_automation_principal_login() } } + + +// ============================================================================================ +// THE DECLARED RUNG DROP for both fleet-global values above (DESIGN 4b(3)). +// ============================================================================================ +// +// WHY ONE ROW AND NOT TWO. The context ceiling and the slot count are the same defect in two +// spellings: a value measured on one node, one artifact, one runtime configuration and one +// concurrency condition, then applied to two hosts and every realization the cell can load. Two +// rows would be two places to update when one restoration discharges both, and the restoration IS +// shared -- neither value becomes qualified until per-realization observation exists. +// +// WHAT WAS ACTUALLY MEASURED, so the gap between evidence and scope is legible rather than implied: +// spark-a3ee, the hf.co/antirez/deepseek-v4-gguf IQ2_XXS build, Ollama 0.32.9, one resident session, +// buffer figures read from /api/ps. That is one cell of a matrix whose other axes are the six other +// resident models, the second host, and every concurrency above one. +// +// THE TRIGGER NAMES THE CAPABILITY AND NOT AN ARTIFACT. A per-realization observation table is not +// sufficient by itself: while a global default exists, an unmeasured realization still INHERITS +// these numbers automatically, so measuring some of the roster leaves the same unqualified +// inheritance for the rest. The capability is that no realization can be served at a configuration +// it has not been observed at -- which requires both the observations and the removal of automatic +// inheritance. +data spark_serving_fleet_global_configuration_drop: GuaranteeRungDrop = GuaranteeRungDrop { + subject: "the serving cell's default context window and concurrent slot count, applied fleet-wide from single-realization evidence", + previous: MechanicallyPreventable, + temporary: Mitigatable, + reason: DeletedWithoutReplacement, + population: [ + "gunbc.spark.serving_desired spark_serving_desired_context_length", + "gunbc.spark.serving_desired spark_serving_desired_serving_slots_value", + ], + restoration_trigger: "no realization can be served at a configuration it has not been observed at: a per-realization runner-memory observation exists for every model the cell loads, at the configuration selected for it, AND the fleet-global default no longer confers an unmeasured realization automatic admission -- an unobserved realization must refuse rather than inherit. An observation table alone does not discharge this row, because inheritance is the half that makes an unmeasured model's admission silent", +} diff --git a/dag/gunbc/spark/serving_install_paths.dag b/dag/gunbc/spark/serving_install_paths.dag index cbcc8517151..4662d49e62f 100644 --- a/dag/gunbc/spark/serving_install_paths.dag +++ b/dag/gunbc/spark/serving_install_paths.dag @@ -56,7 +56,8 @@ type SparkServingUserUnitSpec { exec_subcommand: NonEmptyStr restart: NonEmptyStr wanted_by: NonEmptyStr - configured_context: TokenCount + default_context: TokenCount + serving_slots: Nat } // THE MODEL STORE PATHS, derived from the same models root the unit's OLLAMA_MODELS points at, so @@ -311,14 +312,15 @@ fn spark_serving_user_unit_path() -> FilePath { ) as FilePath } -// `configured_context` is a PARAMETER rather than a literal here for the same reason +// `default_context` is a PARAMETER rather than a literal here for the same reason // `bind_host_port` is: it is desired state, and its authority is the occurrence's launch profile -// (gunbc.spark.serving_desired spark_serving_desired_configured_context). Spelling a number here +// (gunbc.spark.serving_desired spark_serving_desired_default_context). Spelling a number here // would make this module a second authority for the window and would put the declared profile back // out of the unit's reach -- exactly the defect this change closes. fn spark_serving_user_unit_spec( bind_host_port: NonEmptyStr, - configured_context: TokenCount, + default_context: TokenCount, + serving_slots: Nat, ) -> SparkServingUserUnitSpec { SparkServingUserUnitSpec { description: "gunbc Spark serving (Ollama gpt-oss:20b)" as NonEmptyStr, @@ -332,7 +334,8 @@ fn spark_serving_user_unit_spec( exec_subcommand: "serve" as NonEmptyStr, restart: "on-failure" as NonEmptyStr, wanted_by: "default.target" as NonEmptyStr, - configured_context: configured_context, + default_context: default_context, + serving_slots: serving_slots, } } diff --git a/dag/gunbc/spark/serving_membership.dag b/dag/gunbc/spark/serving_membership.dag index 6e1406fab91..697d8a94598 100644 --- a/dag/gunbc/spark/serving_membership.dag +++ b/dag/gunbc/spark/serving_membership.dag @@ -26,7 +26,8 @@ import gunbc.spark.serving_install_paths { import gunbc.spark.serving_desired { spark_serving_desired_bind_host_port, spark_serving_automation_principal_login, - spark_serving_desired_configured_context, + spark_serving_desired_default_context, + spark_serving_desired_serving_slots, } import gunbc.spark.serving_unit_identity { spark_serving_unit_name } import gunbc.spark.serving_unit_render { spark_serving_desired_user_unit_content_identity } @@ -310,7 +311,8 @@ fn spark_serving_user_unit_file_write_intent() -> SparkServingUserUnitWriteInten path: spark_serving_user_unit_path(), spec: spark_serving_user_unit_spec( bind_host_port: spark_serving_desired_bind_host_port() as NonEmptyStr, - configured_context: spark_serving_desired_configured_context(), + default_context: spark_serving_desired_default_context(), + serving_slots: spark_serving_desired_serving_slots(), ), } } diff --git a/dag/gunbc/spark/serving_release.dag b/dag/gunbc/spark/serving_release.dag index 230f97bd571..1727051041d 100644 --- a/dag/gunbc/spark/serving_release.dag +++ b/dag/gunbc/spark/serving_release.dag @@ -83,7 +83,8 @@ type SparkServingRelease { // identity domains. type OllamaServingLaunchProfile { bind: SparkServingBindListen - configured_context: TokenCount + default_context: TokenCount + serving_slots: Nat } // The identity is DERIVED from the release, never authored. It previously carried a NonEmptyStr diff --git a/dag/gunbc/spark/serving_unit_render.dag b/dag/gunbc/spark/serving_unit_render.dag index d3a00da3608..cdebc4a4637 100644 --- a/dag/gunbc/spark/serving_unit_render.dag +++ b/dag/gunbc/spark/serving_unit_render.dag @@ -21,7 +21,7 @@ import extdeps.systemd.unit_file { import extdeps.ollama.server_env { ollama_host_env_assignment, ollama_models_env_assignment, - ollama_context_length_env_assignment, + ollama_context_length_env_assignment, ollama_num_parallel_env_assignment, } import gunbc.spark.serving_install_paths { SparkServingUserUnitSpec, @@ -32,7 +32,8 @@ import gunbc.spark.serving_install_paths { } import gunbc.spark.serving_desired { spark_serving_desired_bind_host_port, - spark_serving_desired_configured_context, + spark_serving_desired_default_context, + spark_serving_desired_serving_slots, } // THE ONE PLACE A SPARK SERVING UNIT BECOMES BYTES. @@ -79,7 +80,10 @@ fn spark_serving_user_unit_file(spec: SparkServingUserUnitSpec) -> SystemdUnitFi assignment: ollama_models_env_assignment(root: spark_serving_models_root() as NonEmptyStr), }, Environment { - assignment: ollama_context_length_env_assignment(configured_context: spec.configured_context), + assignment: ollama_context_length_env_assignment(default_context: spec.default_context), + }, + Environment { + assignment: ollama_num_parallel_env_assignment(slots: spec.serving_slots), }, WorkingDirectory { path: spark_serving_required_runtime_root() as NonEmptyStr }, ExecStart { command: spark_serving_exec_start_command(spec: spec) }, @@ -100,7 +104,8 @@ fn spark_serving_desired_user_unit_text() -> String { spark_serving_user_unit_text( spec: spark_serving_user_unit_spec( bind_host_port: spark_serving_desired_bind_host_port() as NonEmptyStr, - configured_context: spark_serving_desired_configured_context(), + default_context: spark_serving_desired_default_context(), + serving_slots: spark_serving_desired_serving_slots(), ), ) } diff --git a/dag/test/claim/spark/spark_serving_release_witness_test.dag b/dag/test/claim/spark/spark_serving_release_witness_test.dag index 43dc6f1c2b0..b6d9ead3e34 100644 --- a/dag/test/claim/spark/spark_serving_release_witness_test.dag +++ b/dag/test/claim/spark/spark_serving_release_witness_test.dag @@ -80,14 +80,16 @@ fn witness_bind() -> SparkServingBindListen { fn profile_base() -> OllamaServingLaunchProfile { OllamaServingLaunchProfile { bind: witness_bind(), - configured_context: token_count(count: 4096), + default_context: token_count(count: 4096), + serving_slots: 4, } } fn profile_other_window() -> OllamaServingLaunchProfile { OllamaServingLaunchProfile { bind: witness_bind(), - configured_context: token_count(count: 8192), + default_context: token_count(count: 8192), + serving_slots: 4, } } @@ -107,8 +109,8 @@ test fn release_identity_is_derived_not_authored() -> Bool { test fn launch_profile_window_does_not_change_release_identity() -> Bool { digest_of(spec: release_base()) == digest_of(spec: release_base()) - && token_count_value(t: profile_base().configured_context) - != token_count_value(t: profile_other_window().configured_context) + && token_count_value(t: profile_base().default_context) + != token_count_value(t: profile_other_window().default_context) } fn refusal_specimen() -> SparkServingReconciliation { diff --git a/dag/test/claim/spark/spark_serving_unit_render_witness_test.dag b/dag/test/claim/spark/spark_serving_unit_render_witness_test.dag index 12cf6fa9de8..5638962ad58 100644 --- a/dag/test/claim/spark/spark_serving_unit_render_witness_test.dag +++ b/dag/test/claim/spark/spark_serving_unit_render_witness_test.dag @@ -14,7 +14,8 @@ import gunbc.spark.serving_unit_render { } import gunbc.spark.serving_desired { spark_serving_desired_bind_host_port, - spark_serving_desired_configured_context, + spark_serving_desired_default_context, + spark_serving_desired_serving_slots, } // THE CONFIGURED CONTEXT WINDOW IS CAUSAL, AND THESE ARE THE ARMS THAT GO RED IF IT STOPS BEING SO. @@ -32,7 +33,8 @@ import gunbc.spark.serving_desired { fn spec_with_window(window: TokenCount) -> SparkServingUserUnitSpec { spark_serving_user_unit_spec( bind_host_port: spark_serving_desired_bind_host_port() as NonEmptyStr, - configured_context: window, + default_context: window, + serving_slots: spark_serving_desired_serving_slots(), ) } @@ -68,6 +70,6 @@ test fn the_configured_window_changes_the_unit_content_identity() -> Bool { // argument and returns one fixed text. test fn the_desired_unit_is_rendered_at_the_declared_window() -> Bool { spark_serving_desired_user_unit_text() - == text_with_window(window: spark_serving_desired_configured_context()) + == text_with_window(window: spark_serving_desired_default_context()) && spark_serving_desired_user_unit_text() != text_with_window(window: token_count(count: 4096)) } From 670873fe3de56838400c9e78d1df1a7660c64f9b Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 19:47:51 +0000 Subject: [PATCH 2/4] The slot count is a measure that cannot be zero, the render has a discriminating witness, and the figures give way to their instrument Three findings from review 58289, all valid. A BARE Nat CARRIED THE SLOT COUNT into the substrate, when std.measure is the authority for domain quantities. It is now PositiveSlotCount, built over PositiveMeasureCount the way PositiveCelsiusDelta is -- which also answers the second half of that finding, that the old carrier could represent a zero-slot configuration. Zero now has no constructor rather than a validator: a serving cell that serves nothing is not a quieter configuration, it is an absent one, and a caller who wants that removes the cell. THE RENDER HAD NO DISCRIMINATING TEST FOR THE NEW DIRECTIVE. Every existing arm varied the window; the slot count was always the desired value, so OLLAMA_NUM_PARALLEL could have stopped being rendered with the whole suite still green. That is the same condition the window arms were written to prevent, one axis late. Three arms added that hold the window fixed and vary only the slots -- bytes, content identity, and the causal link to the declared value -- each stated in both directions so a renderer ignoring its argument fails the equality half. THE ANNOTATIONS TRANSCRIBED RUNNER FIGURES rather than naming a producer that re-derives them, which DESIGN section 6 forbids for the reason this file already demonstrates: the sentence being replaced cited an /api/ps reading as proof of a value that had never been rendered onto either host, and nothing connected the number to the thing that owned it. Every byte figure is removed. The instrument is named instead -- gunbc.spark.serving_observe spark_serving_observe_ci_wet, whose residency readback in gunbc.spark.serving_readback_parse already decodes the runner's size_vram. The arguments survive as comparisons rather than as copied numbers: the same context_length at one slot and at two with a larger buffer at two still says the count multiplies memory and does not divide context. stage0 mirrors PositiveSlotCount alongside PositiveCelsiusDelta; cargo check and clippy clean. Spark witnesses re-run: unit render including the three new arms, release, observation. --- dag/extdeps/ollama/server_env.dag | 7 +-- dag/gunbc/spark/serving_desired.dag | 60 +++++++++++-------- dag/gunbc/spark/serving_install_paths.dag | 6 +- dag/gunbc/spark/serving_release.dag | 4 +- dag/std/measure.dag | 26 ++++++++ .../spark_serving_release_witness_test.dag | 6 +- ...spark_serving_unit_render_witness_test.dag | 56 ++++++++++++++++- src/v1/stage0/src/std_measure.rs | 28 +++++++++ 8 files changed, 156 insertions(+), 37 deletions(-) diff --git a/dag/extdeps/ollama/server_env.dag b/dag/extdeps/ollama/server_env.dag index c90b674d838..d01c2b7bcd2 100644 --- a/dag/extdeps/ollama/server_env.dag +++ b/dag/extdeps/ollama/server_env.dag @@ -1,8 +1,7 @@ module extdeps.ollama.server_env import std.types { NonEmptyStr, String } -import std.nat { Nat } -import std.measure { TokenCount, token_count_value } +import std.measure { TokenCount, token_count_value, PositiveSlotCount, positive_slot_count_value } import extdeps.external_authority { ExternalAuthority, ExternalModelScope, ExternalSubjectRef } import extdeps.uri { Uri, Https } import std.decl_ref { DeclarationRef, WholeDeclaration } @@ -115,9 +114,9 @@ fn ollama_context_length_env_assignment(default_context: TokenCount) -> NonEmpty // and the per-slot price is a property of the REALIZATION -- a compressed-cache architecture pays // little per slot and an uncompressed one pays a great deal. A slot count chosen against one model's // per-slot cost is not evidence for another's. -fn ollama_num_parallel_env_assignment(slots: Nat) -> NonEmptyStr { +fn ollama_num_parallel_env_assignment(slots: PositiveSlotCount) -> NonEmptyStr { ollama_server_env_assignment( variable: OllamaNumParallel, - value: to_string(slots) as NonEmptyStr, + value: to_string(positive_slot_count_value(slots: slots)) as NonEmptyStr, ) } diff --git a/dag/gunbc/spark/serving_desired.dag b/dag/gunbc/spark/serving_desired.dag index 2d200e71901..7c5eda6985a 100644 --- a/dag/gunbc/spark/serving_desired.dag +++ b/dag/gunbc/spark/serving_desired.dag @@ -1,8 +1,9 @@ module gunbc.spark.serving_desired import std.types { String, Bool, List, NonEmptyStr } -import std.measure { TokenCount, token_count } -import std.nat { Nat } +import std.measure { + TokenCount, token_count, PositiveSlotCount, positive_slot_count, positive_measure_count, +} import gunbc.guarantee_rung_drop { GuaranteeRungDrop, MechanicallyPreventable, Mitigatable, DeletedWithoutReplacement, } @@ -251,7 +252,8 @@ fn spark_serving_desired_launch_profile() -> OllamaServingLaunchProfile { // TWO VALUES STOOD HERE BEFORE AND NEITHER WAS EVER ENACTED. The 8192 was never rendered into the // unit at all. The 131072 that replaced it was rendered by this module's own renderer and still did // not reach either host: read directly on 2026-09-01, the deployed user unit on spark-a3ee and -// spark-3bd5 carried three Environment lines and no OLLAMA_CONTEXT_LENGTH among them. +// spark-3bd5 carried three Environment lines and no OLLAMA_CONTEXT_LENGTH among them. That absence +// is what the observed-versus-desired unit comparison exists to report, and did not. // // THE EVIDENCE SENTENCE IS WHY THAT WENT UNNOTICED FOR SO LONG, and it is the more useful half of // this history. It read that the wet observation was EVIDENCE and not source, citing both hosts @@ -261,10 +263,16 @@ fn spark_serving_desired_launch_profile() -> OllamaServingLaunchProfile { // could not have discriminated: with no assignment on either host the number being confirmed was the // runtime's own automatic default. A declaration cited its own absence as confirmation. // -// 1048576 IS SELECTED, and what makes it defensible is a load rather than a reading: the served -// build declares `deepseek4.context_length` 1048576 and at that full window its runner reports -// 88,865,253,620 bytes of buffer. Re-derive by loading at an explicit num_ctx and reading /api/ps -// back; that instrument, not this number, is the authority for what any build costs. +// 1048576 IS SELECTED, and what makes it defensible is a LOAD rather than a reading: the served +// build declares its full window, and at that window it loads and serves within the cell's memory. +// +// NO BYTE FIGURE IS TRANSCRIBED HERE, and that is deliberate rather than an omission. A number +// copied into prose is unreachable from the thing that owns it and rots without either end being +// touched -- which is exactly how the sentence this replaces came to cite an /api/ps reading as +// proof of a value that had never been rendered onto either host. The instrument is +// `gunbc.spark.serving_observe spark_serving_observe_ci_wet`, whose residency readback is parsed by +// `gunbc.spark.serving_readback_parse` and already decodes the runner's size_vram figure; that +// producer, not any figure written here, is the authority for what a build costs at a window. // // A smaller default remains writable. A larger one than a model declares is a real request the // runtime is free to refuse -- Ollama clamps to the model's declared maximum, so this is a CEILING @@ -274,31 +282,35 @@ data spark_serving_desired_context_length: TokenCount = token_count(count: 10485 // HOW MANY REQUESTS ONE LOADED MODEL SERVES CONCURRENTLY -- the P1b row, now authored. // // THE NOTE ABOVE SAID THIS TRADED AGAINST CONTEXT "BECAUSE THE SLOT COUNT DIVIDES THE CONTEXT -// WINDOW". That sentence is deleted rather than softened, because it is false. Measured on one host, -// one model, changing only OLLAMA_NUM_PARALLEL: at one slot the runner reported context_length -// 1048576 and 88,865,253,620 bytes of buffer; at two slots it reported context_length 1048576 and -// 90,543,761,652. Each slot receives its own full window. THE COUNT MULTIPLIES MEMORY; IT DOES NOT -// DIVIDE CONTEXT. The two axes do trade inside one budget, but multiplicatively. +// WINDOW". That sentence is deleted rather than softened, because it is false. Holding the model and +// the host fixed and varying only OLLAMA_NUM_PARALLEL, the runner reports the SAME context_length at +// one slot and at two, and a larger buffer at two. Each slot receives its own full window: THE COUNT +// MULTIPLIES MEMORY, IT DOES NOT DIVIDE CONTEXT. The two axes do trade inside one budget, but +// multiplicatively. +// +// Re-derive with the same instrument named below rather than from figures written here -- the +// readback already decodes both quantities the comparison needs, and a transcribed pair would be a +// second copy of a fact whose producer exists. // -// WHY THE VALUE IS NOT 1. Unset, the runtime chose a single slot, and a single slot SERIALIZES: on -// 2026-09-01 a ~61,000-token prefill occupied one host for roughly two hundred seconds while every -// other caller queued, and the queue was read as a latency regression in the model rather than as -// contention. That is not a throughput cost, it is a head-of-line block, and the harness workload -// this cell exists to serve is concurrent by construction. +// WHY THE VALUE IS NOT 1. Unset, the runtime chose a single slot, and a single slot SERIALIZES: a +// long prefill occupies the host for its whole duration while every other caller queues behind it, +// and the queue is readable as a latency regression in the model rather than as contention -- which +// is how it was in fact first read. That is not a throughput cost, it is a head-of-line block, and +// the harness workload this cell exists to serve is concurrent by construction. // // WHAT THIS VALUE IS NOT EVIDENCE FOR, and the reason a rung drop is filed below rather than a -// footnote written here: the per-slot price is a property of the REALIZATION. The measured -// 1,678,508,032 bytes is a compressed-cache figure for one DeepSeek build; an architecture without -// that compression pays far more per slot, and llama4:scout in particular declares a window far -// beyond this ceiling with no such compression. A count chosen against one model's per-slot cost is -// not evidence for another's, and this row applies to every realization the cell loads. -data spark_serving_desired_serving_slots_value: Nat = 4 +// footnote written here: the per-slot price is a property of the REALIZATION. It was measured on a +// build whose cache is compressed, and an architecture without that compression pays far more per +// slot -- llama4:scout in particular declares a window far beyond this ceiling and has no such +// compression. A count chosen against one model's per-slot cost is not evidence for another's, and +// this row applies to every realization the cell loads. +data spark_serving_desired_serving_slots_value: PositiveSlotCount = positive_slot_count(count: positive_measure_count(predecessor: 3)) // The unit's window is DERIVED from the serving policy rather than authored beside it. This is the // causal link the repair adds: the policy is the single authority, the unit text is a function of // it, and the unit's content identity is a function of the text -- so a policy change cannot fail // to reach the file, and cannot reach it without producing a member delta. -fn spark_serving_desired_serving_slots() -> Nat { +fn spark_serving_desired_serving_slots() -> PositiveSlotCount { spark_serving_desired_launch_profile().serving_slots } diff --git a/dag/gunbc/spark/serving_install_paths.dag b/dag/gunbc/spark/serving_install_paths.dag index 4662d49e62f..342dab9df2a 100644 --- a/dag/gunbc/spark/serving_install_paths.dag +++ b/dag/gunbc/spark/serving_install_paths.dag @@ -1,7 +1,7 @@ module gunbc.spark.serving_install_paths import std.types { String, FilePath, NonEmptyStr } -import std.measure { TokenCount } +import std.measure { TokenCount, PositiveSlotCount } import std.content_hash { Fnv1a64Structural } import gunbc.spark.serving_runtime_receipt { spark_serving_executor_home } import gunbc.spark.serving_unit_identity { spark_serving_unit_name } @@ -57,7 +57,7 @@ type SparkServingUserUnitSpec { restart: NonEmptyStr wanted_by: NonEmptyStr default_context: TokenCount - serving_slots: Nat + serving_slots: PositiveSlotCount } // THE MODEL STORE PATHS, derived from the same models root the unit's OLLAMA_MODELS points at, so @@ -320,7 +320,7 @@ fn spark_serving_user_unit_path() -> FilePath { fn spark_serving_user_unit_spec( bind_host_port: NonEmptyStr, default_context: TokenCount, - serving_slots: Nat, + serving_slots: PositiveSlotCount, ) -> SparkServingUserUnitSpec { SparkServingUserUnitSpec { description: "gunbc Spark serving (Ollama gpt-oss:20b)" as NonEmptyStr, diff --git a/dag/gunbc/spark/serving_release.dag b/dag/gunbc/spark/serving_release.dag index 1727051041d..7545664e9b4 100644 --- a/dag/gunbc/spark/serving_release.dag +++ b/dag/gunbc/spark/serving_release.dag @@ -7,7 +7,7 @@ import gunbc.model_artifact { ModelArtifact } import extdeps.ollama.api { OllamaModelRef, ollama_model_ref_wire, ollama_default_port } import v2.std.algebra { fold_list } import std.nat { Nat } -import std.measure { TokenCount } +import std.measure { TokenCount, PositiveSlotCount } import gunbc.spark.serving_realization { SystemPrincipal, ServingDeploymentTarget, UserUnitRealization, EscalationShortfall, NoNewPrivilegeRequired, NewPrivilegeRequired, ShortfallUndecidable, @@ -84,7 +84,7 @@ type SparkServingRelease { type OllamaServingLaunchProfile { bind: SparkServingBindListen default_context: TokenCount - serving_slots: Nat + serving_slots: PositiveSlotCount } // The identity is DERIVED from the release, never authored. It previously carried a NonEmptyStr diff --git a/dag/std/measure.dag b/dag/std/measure.dag index 6b52e6e5c54..330a40de0fd 100644 --- a/dag/std/measure.dag +++ b/dag/std/measure.dag @@ -556,6 +556,17 @@ fn positive_measure_count_value( type PositiveCelsiusDelta = PositiveCelsiusDeltaValue { count: PositiveMeasureCount } +// A count of CONCURRENT SERVING SLOTS -- how many requests one loaded model answers at once. +// +// It is on the Count axis, a sibling of HardwareThreadCount, CharacterCount and TokenCount with a +// different thing being counted, so it takes only arguments std.measure already owns. What it does +// NOT take is Nat, and the difference is the point: a slot count of zero describes a serving cell +// that serves nothing, which is not a quieter configuration but an absent one. Built over +// PositiveMeasureCount the way PositiveCelsiusDelta is, so zero has no constructor rather than a +// validator -- and a caller who wants "off" must remove the cell rather than set it to nothing. +type PositiveSlotCount + = PositiveSlotCountValue { count: PositiveMeasureCount } + type RevolutionsPerMinute = Measure // A per-minute count of discrete events -- a frequency-family rate whose period marker is @@ -675,6 +686,21 @@ fn celsius_delta_count(delta: CelsiusDelta) -> Int { measure_count(delta) } +fn positive_slot_count( + count: PositiveMeasureCount, +) -> PositiveSlotCount { + PositiveSlotCountValue { count: count } +} + +fn positive_slot_count_value( + slots: PositiveSlotCount, +) -> Int { + match slots { + PositiveSlotCountValue { count: count } => + positive_measure_count_value(count: count) + } +} + fn positive_celsius_delta( count: PositiveMeasureCount, ) -> PositiveCelsiusDelta { diff --git a/dag/test/claim/spark/spark_serving_release_witness_test.dag b/dag/test/claim/spark/spark_serving_release_witness_test.dag index b6d9ead3e34..72c43a428cd 100644 --- a/dag/test/claim/spark/spark_serving_release_witness_test.dag +++ b/dag/test/claim/spark/spark_serving_release_witness_test.dag @@ -1,7 +1,7 @@ module test.claim.spark_serving_release_witness_test import std.types { String, Bool, List, Int, NonEmptyStr } -import std.measure { token_count, token_count_value } +import std.measure { token_count, token_count_value, positive_slot_count, positive_measure_count } import std.content_hash { content_hash_of_value, ContentHash } import product.placement_supply { HostIdentity } import gunbc.spark.serving_realization { @@ -81,7 +81,7 @@ fn profile_base() -> OllamaServingLaunchProfile { OllamaServingLaunchProfile { bind: witness_bind(), default_context: token_count(count: 4096), - serving_slots: 4, + serving_slots: positive_slot_count(count: positive_measure_count(predecessor: 3)), } } @@ -89,7 +89,7 @@ fn profile_other_window() -> OllamaServingLaunchProfile { OllamaServingLaunchProfile { bind: witness_bind(), default_context: token_count(count: 8192), - serving_slots: 4, + serving_slots: positive_slot_count(count: positive_measure_count(predecessor: 3)), } } diff --git a/dag/test/claim/spark/spark_serving_unit_render_witness_test.dag b/dag/test/claim/spark/spark_serving_unit_render_witness_test.dag index 5638962ad58..074060f59a6 100644 --- a/dag/test/claim/spark/spark_serving_unit_render_witness_test.dag +++ b/dag/test/claim/spark/spark_serving_unit_render_witness_test.dag @@ -1,7 +1,10 @@ module test.claim.spark_serving_unit_render_witness_test import std.types { Bool, String, NonEmptyStr } -import std.measure { TokenCount, token_count } +import std.measure { + TokenCount, token_count, PositiveSlotCount, positive_slot_count, positive_measure_count, +} +import std.nat { Nat } import gunbc.spark.serving_install_paths { SparkServingUserUnitSpec, spark_serving_user_unit_spec, @@ -38,6 +41,22 @@ fn spec_with_window(window: TokenCount) -> SparkServingUserUnitSpec { ) } +fn spec_with_slots(slots: PositiveSlotCount) -> SparkServingUserUnitSpec { + spark_serving_user_unit_spec( + bind_host_port: spark_serving_desired_bind_host_port() as NonEmptyStr, + default_context: spark_serving_desired_default_context(), + serving_slots: slots, + ) +} + +fn text_with_slots(slots: PositiveSlotCount) -> String { + spark_serving_user_unit_text(spec: spec_with_slots(slots: slots)) +} + +fn slots(n: Nat) -> PositiveSlotCount { + positive_slot_count(count: positive_measure_count(predecessor: n)) +} + fn text_with_window(window: TokenCount) -> String { spark_serving_user_unit_text(spec: spec_with_window(window: window)) } @@ -73,3 +92,38 @@ test fn the_desired_unit_is_rendered_at_the_declared_window() -> Bool { == text_with_window(window: spark_serving_desired_default_context()) && spark_serving_desired_user_unit_text() != text_with_window(window: token_count(count: 4096)) } + + +// CONTROL 4: THE SLOT COUNT REACHES THE BYTES, and it is stated separately from the window rather +// than folded into a spec that varies both. A witness that changes two fields at once cannot tell +// which one the renderer consumed, so this arm holds the window at the declared value and varies +// ONLY serving_slots. Without it the OLLAMA_NUM_PARALLEL directive could stop being rendered and +// every other claim in this file would stay green -- which is precisely the condition the window +// arms were written to prevent for the window, one axis late. +// +// Stated in both directions, so a renderer that ignores its argument and returns one fixed text +// fails the equality half rather than passing the inequality half by accident. +test fn the_slot_count_changes_the_unit_bytes() -> Bool { + text_with_slots(slots: slots(n: 0)) != text_with_slots(slots: slots(n: 3)) + && text_with_slots(slots: slots(n: 3)) == text_with_slots(slots: slots(n: 3)) +} + +// CONTROL 5: the member identity moves with the slot count too. The write and observe sides compare +// content identities, so a slot change that reached the text but not the identity would converge as +// a no-op against a host serving a different concurrency. +test fn the_slot_count_changes_the_unit_content_identity() -> Bool { + spark_serving_user_unit_content_identity( + wire_base64: spark_serving_user_unit_content_wire(text: text_with_slots(slots: slots(n: 0))), + ) != spark_serving_user_unit_content_identity( + wire_base64: spark_serving_user_unit_content_wire(text: text_with_slots(slots: slots(n: 3))), + ) +} + +// CONTROL 6: the causal link for the concurrency axis. The desired unit is rendered at the slot +// count the serving policy declares and at no other, so the policy stays the single authority the +// bytes derive from. +test fn the_desired_unit_is_rendered_at_the_declared_slot_count() -> Bool { + spark_serving_desired_user_unit_text() + == text_with_slots(slots: spark_serving_desired_serving_slots()) + && spark_serving_desired_user_unit_text() != text_with_slots(slots: slots(n: 0)) +} diff --git a/src/v1/stage0/src/std_measure.rs b/src/v1/stage0/src/std_measure.rs index 4716f07ee13..f9968ace468 100644 --- a/src/v1/stage0/src/std_measure.rs +++ b/src/v1/stage0/src/std_measure.rs @@ -5,6 +5,7 @@ use self::ClockBasis::*; use self::ClockDomain::*; use self::InstantOrder::*; use self::PositiveCelsiusDelta::*; +use self::PositiveSlotCount::*; use self::PositiveMeasureCount::*; use self::PositiveMeasureCountBuild::*; use self::PositiveMillisecond::*; @@ -559,6 +560,19 @@ pub fn positive_measure_count_value(count: Rc) -> i64 { } } +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +#[serde(tag = "_variant")] +pub enum PositiveSlotCount { + PositiveSlotCountValue { count: Rc }, +} +impl PositiveSlotCount { + pub fn count(&self) -> Rc { + match self { + PositiveSlotCount::PositiveSlotCountValue { count: __val, .. } => __val.clone(), + } + } +} + #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(tag = "_variant")] pub enum PositiveCelsiusDelta { @@ -716,6 +730,20 @@ pub fn celsius_delta_count(delta: CelsiusDelta) -> i64 { measure_count(delta.clone()) } +pub fn positive_slot_count(count: Rc) -> Rc { + Rc::new(PositiveSlotCount::PositiveSlotCountValue { + count: count.clone(), + }) +} + +pub fn positive_slot_count_value(slots: Rc) -> i64 { + match (*slots.clone()).clone() { + PositiveSlotCount::PositiveSlotCountValue { count: count, .. } => { + positive_measure_count_value(count.clone()) + } + } +} + pub fn positive_celsius_delta(count: Rc) -> Rc { Rc::new(PositiveCelsiusDelta::PositiveCelsiusDeltaValue { count: count.clone(), From b1b8775996df216993fdd8ed4a7de5df94851982 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 20:20:54 +0000 Subject: [PATCH 3/4] Install the regen actuator's own mirror bytes instead of hand-written ones required-witnesses-build failed on 670873f. The cause is not the modeling: it is that I hand-wrote src/v1/stage0/src/std_measure.rs, a file whose header says "Generated by v1 compiler -- do not edit." The generator places a new coproduct in the .dag's own declaration order and sorts the `use` lines; I inserted PositiveSlotCount ahead of PositiveCelsiusDelta in both places. Same types, same functions, different bytes -- and the drift gate compares bytes, which is the entire point of it. Running claim_executor --required-regen --source-root dag --source-root src/v2 names std_measure.rs among the divergent surfaces, and its candidate differs from what I wrote by exactly those two reorderings and nothing else. The candidate is installed verbatim. I HAVE NOW MADE THIS MISTAKE TWICE IN ONE DAY. 2017773c63c regenerated this same mirror for TokensPerSecond and its message recorded why local verification cannot catch it -- every witness resolves .dag sources through an already-built binary, so the .dag change is live and the seed the next build needs is not. Adding a type to std/measure.dag has a mandatory second half, and knowing that is not the same as doing it. The actuator is cheap; running it is the check. --- src/v1/stage0/src/std_measure.rs | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/src/v1/stage0/src/std_measure.rs b/src/v1/stage0/src/std_measure.rs index f9968ace468..71c9f0920d5 100644 --- a/src/v1/stage0/src/std_measure.rs +++ b/src/v1/stage0/src/std_measure.rs @@ -5,10 +5,10 @@ use self::ClockBasis::*; use self::ClockDomain::*; use self::InstantOrder::*; use self::PositiveCelsiusDelta::*; -use self::PositiveSlotCount::*; use self::PositiveMeasureCount::*; use self::PositiveMeasureCountBuild::*; use self::PositiveMillisecond::*; +use self::PositiveSlotCount::*; use self::Quantity::*; use self::Scale::*; pub use crate::extdeps_currency_currency::CurrencyCode; @@ -562,26 +562,26 @@ pub fn positive_measure_count_value(count: Rc) -> i64 { #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(tag = "_variant")] -pub enum PositiveSlotCount { - PositiveSlotCountValue { count: Rc }, +pub enum PositiveCelsiusDelta { + PositiveCelsiusDeltaValue { count: Rc }, } -impl PositiveSlotCount { +impl PositiveCelsiusDelta { pub fn count(&self) -> Rc { match self { - PositiveSlotCount::PositiveSlotCountValue { count: __val, .. } => __val.clone(), + PositiveCelsiusDelta::PositiveCelsiusDeltaValue { count: __val, .. } => __val.clone(), } } } #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(tag = "_variant")] -pub enum PositiveCelsiusDelta { - PositiveCelsiusDeltaValue { count: Rc }, +pub enum PositiveSlotCount { + PositiveSlotCountValue { count: Rc }, } -impl PositiveCelsiusDelta { +impl PositiveSlotCount { pub fn count(&self) -> Rc { match self { - PositiveCelsiusDelta::PositiveCelsiusDeltaValue { count: __val, .. } => __val.clone(), + PositiveSlotCount::PositiveSlotCountValue { count: __val, .. } => __val.clone(), } } } From ac9ff995d94e7a82fc838b3e5d3e5552c1ddcdc7 Mon Sep 17 00:00:00 2001 From: gunbc-ci-auto-heal Date: Tue, 1 Sep 2026 21:30:19 +0000 Subject: [PATCH 4/4] Name the instrument for the slot/context comparison instead of transcribing its figures MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review 58311 on #9960. The note carried two byte figures and two context_length readings copied from a run, which DESIGN ยง6 forbids for the reason the neighbouring serving_desired.dag note already states: a number copied into prose is unreachable from the producer that owns it and rots without either end being touched. The qualitative conclusion is what the authority needs -- same context_length at one slot and at two, a larger buffer at two -- and the readback producer is named so the comparison is re-derived rather than read. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01U397y4s3dSBof7vGPAX87G --- dag/extdeps/ollama/server_env.dag | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/dag/extdeps/ollama/server_env.dag b/dag/extdeps/ollama/server_env.dag index d01c2b7bcd2..e875d5b3cbb 100644 --- a/dag/extdeps/ollama/server_env.dag +++ b/dag/extdeps/ollama/server_env.dag @@ -105,10 +105,16 @@ fn ollama_context_length_env_assignment(default_context: TokenCount) -> NonEmpty // duration, which is not a throughput cost but a queue. // // THE SLOT COUNT DOES NOT DIVIDE THE CONTEXT WINDOW. That is worth stating in the authority because -// the opposite was written down and believed. Measured on one host, one model, changing only this -// variable: at one slot the runner reported context_length 1048576 and 88,865,253,620 bytes; at two -// slots it reported context_length 1048576 and 90,543,761,652 bytes. Each slot receives its OWN full -// window and costs its own cache. The count multiplies memory; it does not divide context. +// the opposite was written down and believed. Holding the host and the model fixed and varying only +// this variable, the runner reports the SAME context_length at one slot and at two, and a larger +// buffer at two: each slot receives its OWN full window and costs its own cache. The count +// multiplies memory; it does not divide context. +// +// NO FIGURE IS TRANSCRIBED HERE. Re-derive the comparison with the instrument that owns it -- +// `gunbc.spark.serving_observe spark_serving_observe_ci_wet`, whose residency readback is decoded by +// `gunbc.spark.serving_readback_parse` and already yields both the context_length and the size_vram +// the comparison needs. A pair of numbers copied into this comment would be unreachable from that +// producer and would rot without either end being touched. // // So the two axes do trade inside one memory budget, but multiplicatively rather than by division, // and the per-slot price is a property of the REALIZATION -- a compressed-cache architecture pays