Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
63 changes: 50 additions & 13 deletions dag/extdeps/ollama/server_env.dag
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
module extdeps.ollama.server_env

import std.types { NonEmptyStr, String }
import std.measure { TokenCount, token_count_value }
import std.measure { TokenCount, token_count_value, PositiveSlotCount, positive_slot_count_value }
import extdeps.external_authority { ExternalAuthority, ExternalModelScope, ExternalSubjectRef }
import extdeps.uri { Uri, Https }
import std.decl_ref { DeclarationRef, WholeDeclaration }
Expand Down Expand Up @@ -44,12 +44,14 @@ type OllamaServerEnvVariable
= OllamaHost
| OllamaModels
| OllamaContextLength
| OllamaNumParallel

fn ollama_server_env_name(variable: OllamaServerEnvVariable) -> NonEmptyStr {
match variable {
OllamaHost => "OLLAMA_HOST" as NonEmptyStr
OllamaModels => "OLLAMA_MODELS" as NonEmptyStr
OllamaContextLength => "OLLAMA_CONTEXT_LENGTH" as NonEmptyStr
OllamaNumParallel => "OLLAMA_NUM_PARALLEL" as NonEmptyStr
}
}

Expand All @@ -71,21 +73,56 @@ fn ollama_models_env_assignment(root: NonEmptyStr) -> NonEmptyStr {
ollama_server_env_assignment(variable: OllamaModels, value: root)
}

// The context window the SERVER allocates for a loaded model, in tokens.
// THE SERVER'S DEFAULT CONTEXT WINDOW, in tokens -- a default, not a limit, and the distinction is
// this function's whole subject.
//
// The server decides what is allocated; the model file only declares what it was trained to
// support. So a deployment that wants a model's full declared window has to say so, and one that
// says nothing gets Ollama's own default rather than the model's declaration. Which window a given
// deployment wants is a deployment fact, so it is a parameter here and never a default -- the same
// reason `ollama_host_env_assignment` refuses to spell a bind address.
// WHAT THIS SAID BEFORE, AND WHY IT WAS A MEANING FORK. It read that this is "the context window the
// SERVER allocates" and "the CONFIGURED limit". Upstream 0.32.9 resolves options in layers -- the
// environment-derived value first, then the model's own options, then the request's -- and an
// explicit num_ctx at either later layer REPLACES this one. So it never limited anything: it is the
// value the server uses when no model or request override applies. Calling it a limit gave one name
// two materially different contracts, and a deployment reading it as a ceiling would believe it had
// bounded something a caller can raise at will.
//
// This is the CONFIGURED limit, which is not the EFFECTIVE one. What a loaded model actually
// allocated is a separate observation read back from the running server; a request the host cannot
// satisfy is resolved by the runtime, not by this assignment, so the two are different facts and
// must not be collapsed into one.
fn ollama_context_length_env_assignment(configured_context: TokenCount) -> NonEmptyStr {
// Which default a deployment wants is a deployment fact, so it is a parameter here and never a
// default of ours -- the same reason `ollama_host_env_assignment` refuses to spell a bind address.
//
// THE DEFAULT IS NOT THE EFFECTIVE VALUE, and that separation is what a reader most needs. What a
// loaded runner actually holds is `/api/ps.context_length`, a distinct observation with its own
// instrument. This assignment cannot be evidence for that reading, and that reading cannot be
// evidence for this assignment -- a measured 2026-09-01 case had both hosts serving a value this
// assignment had never been rendered onto them at all, and an /api/ps reading was cited in the
// consumer as confirmation of a configured value it could not have discriminated.
fn ollama_context_length_env_assignment(default_context: TokenCount) -> NonEmptyStr {
ollama_server_env_assignment(
variable: OllamaContextLength,
value: to_string(token_count_value(t: configured_context)) as NonEmptyStr,
value: to_string(token_count_value(t: default_context)) as NonEmptyStr,
)
}

// HOW MANY REQUESTS ONE LOADED MODEL SERVES CONCURRENTLY. Unset, Ollama chooses for itself, and a
// host that chose one slot SERIALIZES: a long prefill blocks every other caller for its whole
// duration, which is not a throughput cost but a queue.
//
// THE SLOT COUNT DOES NOT DIVIDE THE CONTEXT WINDOW. That is worth stating in the authority because
// the opposite was written down and believed. Holding the host and the model fixed and varying only
// this variable, the runner reports the SAME context_length at one slot and at two, and a larger
// buffer at two: each slot receives its OWN full window and costs its own cache. The count
// multiplies memory; it does not divide context.
//
// NO FIGURE IS TRANSCRIBED HERE. Re-derive the comparison with the instrument that owns it --
// `gunbc.spark.serving_observe spark_serving_observe_ci_wet`, whose residency readback is decoded by
// `gunbc.spark.serving_readback_parse` and already yields both the context_length and the size_vram
// the comparison needs. A pair of numbers copied into this comment would be unreachable from that
// producer and would rot without either end being touched.
//
// So the two axes do trade inside one memory budget, but multiplicatively rather than by division,
// and the per-slot price is a property of the REALIZATION -- a compressed-cache architecture pays
// little per slot and an uncompressed one pays a great deal. A slot count chosen against one model's
// per-slot cost is not evidence for another's.
fn ollama_num_parallel_env_assignment(slots: PositiveSlotCount) -> NonEmptyStr {
ollama_server_env_assignment(
variable: OllamaNumParallel,
value: to_string(positive_slot_count_value(slots: slots)) as NonEmptyStr,
)
}
121 changes: 104 additions & 17 deletions dag/gunbc/spark/serving_desired.dag
Original file line number Diff line number Diff line change
@@ -1,7 +1,12 @@
module gunbc.spark.serving_desired

import std.types { String, Bool, List, NonEmptyStr }
import std.measure { TokenCount, token_count }
import std.measure {
TokenCount, token_count, PositiveSlotCount, positive_slot_count, positive_measure_count,
}
import gunbc.guarantee_rung_drop {
GuaranteeRungDrop, MechanicallyPreventable, Mitigatable, DeletedWithoutReplacement,
}
import product.placement_supply { HostIdentity }
import gunbc.spark.cell_role { spark_serving_cell_hosts }
import gunbc.ollama_runtime_bundle { ollama_runtime_materialization_identity_for_required_release }
Expand Down Expand Up @@ -237,32 +242,80 @@ data spark_serving_fleet_desired: SparkServingFleetDesired = spark_serving_fleet
fn spark_serving_desired_launch_profile() -> OllamaServingLaunchProfile {
OllamaServingLaunchProfile {
bind: spark_serving_desired_bind_listen,
configured_context: spark_serving_desired_context_length,
default_context: spark_serving_desired_context_length,
serving_slots: spark_serving_desired_serving_slots_value,
}
}

// THE CONFIGURED CONTEXT WINDOW, AUTHORED AS THE MIGRATION DECISION.
// THE SERVER'S DEFAULT CONTEXT WINDOW, AUTHORED AS THE MIGRATION DECISION.
//
// TWO VALUES STOOD HERE BEFORE AND NEITHER WAS EVER ENACTED. The 8192 was never rendered into the
// unit at all. The 131072 that replaced it was rendered by this module's own renderer and still did
// not reach either host: read directly on 2026-09-01, the deployed user unit on spark-a3ee and
// spark-3bd5 carried three Environment lines and no OLLAMA_CONTEXT_LENGTH among them. That absence
// is what the observed-versus-desired unit comparison exists to report, and did not.
//
// THE EVIDENCE SENTENCE IS WHY THAT WENT UNNOTICED FOR SO LONG, and it is the more useful half of
// this history. It read that the wet observation was EVIDENCE and not source, citing both hosts
// reporting `context_length: 131072` from /api/ps. But /api/ps reports what a LOADED RUNNER holds,
// which extdeps.ollama.server_env states is a different fact from the server default and must not be
// collapsed into it. An effective reading was admitted as evidence for a configured value, and it
// could not have discriminated: with no assignment on either host the number being confirmed was the
// runtime's own automatic default. A declaration cited its own absence as confirmation.
//
// 1048576 IS SELECTED, and what makes it defensible is a LOAD rather than a reading: the served
// build declares its full window, and at that window it loads and serves within the cell's memory.
//
// NO BYTE FIGURE IS TRANSCRIBED HERE, and that is deliberate rather than an omission. A number
// copied into prose is unreachable from the thing that owns it and rots without either end being
// touched -- which is exactly how the sentence this replaces came to cite an /api/ps reading as
// proof of a value that had never been rendered onto either host. The instrument is
// `gunbc.spark.serving_observe spark_serving_observe_ci_wet`, whose residency readback is parsed by
// `gunbc.spark.serving_readback_parse` and already decodes the runner's size_vram figure; that
// producer, not any figure written here, is the authority for what a build costs at a window.
//
// A smaller default remains writable. A larger one than a model declares is a real request the
// runtime is free to refuse -- Ollama clamps to the model's declared maximum, so this is a CEILING
// ACROSS THE ROSTER and never a per-model promise.
data spark_serving_desired_context_length: TokenCount = token_count(count: 1048576)

// HOW MANY REQUESTS ONE LOADED MODEL SERVES CONCURRENTLY -- the P1b row, now authored.
//
// THE NOTE ABOVE SAID THIS TRADED AGAINST CONTEXT "BECAUSE THE SLOT COUNT DIVIDES THE CONTEXT
// WINDOW". That sentence is deleted rather than softened, because it is false. Holding the model and
// the host fixed and varying only OLLAMA_NUM_PARALLEL, the runner reports the SAME context_length at
// one slot and at two, and a larger buffer at two. Each slot receives its own full window: THE COUNT
// MULTIPLIES MEMORY, IT DOES NOT DIVIDE CONTEXT. The two axes do trade inside one budget, but
// multiplicatively.
//
// The 8192 that stood here was never enacted: nothing rendered the capacity policy into the unit,
// so the declaration sat beside a process it did not configure. Making the carrier causal without
// reselecting the value would therefore not have preserved behavior -- it would have SHRUNK the
// live window 16x on the first converge, as a side effect of a repair whose subject was causality
// and not capacity.
// Re-derive with the same instrument named below rather than from figures written here -- the
// readback already decodes both quantities the comparison needs, and a transcribed pair would be a
// second copy of a fact whose producer exists.
//
// 131072 is SELECTED here, and the wet observation is its EVIDENCE, not its source: both hosts were
// read reporting `context_length: 131072` from the Ollama /api/ps surface, which establishes that
// this value preserves what the running service already exposes. Observed state does not silently
// become desired state -- an operator chose this number and the reading is why it is defensible. A
// smaller window remains writable, and a larger one than the model declares is a real request the
// runtime is free to refuse.
data spark_serving_desired_context_length: TokenCount = token_count(count: 131072)
// WHY THE VALUE IS NOT 1. Unset, the runtime chose a single slot, and a single slot SERIALIZES: a
// long prefill occupies the host for its whole duration while every other caller queues behind it,
// and the queue is readable as a latency regression in the model rather than as contention -- which
// is how it was in fact first read. That is not a throughput cost, it is a head-of-line block, and
// the harness workload this cell exists to serve is concurrent by construction.
//
// WHAT THIS VALUE IS NOT EVIDENCE FOR, and the reason a rung drop is filed below rather than a
// footnote written here: the per-slot price is a property of the REALIZATION. It was measured on a
// build whose cache is compressed, and an architecture without that compression pays far more per
// slot -- llama4:scout in particular declares a window far beyond this ceiling and has no such
// compression. A count chosen against one model's per-slot cost is not evidence for another's, and
// this row applies to every realization the cell loads.
data spark_serving_desired_serving_slots_value: PositiveSlotCount = positive_slot_count(count: positive_measure_count(predecessor: 3))

// The unit's window is DERIVED from the serving policy rather than authored beside it. This is the
// causal link the repair adds: the policy is the single authority, the unit text is a function of
// it, and the unit's content identity is a function of the text -- so a policy change cannot fail
// to reach the file, and cannot reach it without producing a member delta.
fn spark_serving_desired_configured_context() -> TokenCount {
spark_serving_desired_launch_profile().configured_context
fn spark_serving_desired_serving_slots() -> PositiveSlotCount {
spark_serving_desired_launch_profile().serving_slots
}

fn spark_serving_desired_default_context() -> TokenCount {
spark_serving_desired_launch_profile().default_context
}

fn spark_serving_terminal_intent() -> SparkServingIntent {
Expand All @@ -283,3 +336,37 @@ fn spark_serving_automation_principal_login() -> NonEmptyStr {
fn spark_serving_automation_principal() -> NamedPrincipal {
NamedPrincipal { login: spark_serving_automation_principal_login() }
}


// ============================================================================================
// THE DECLARED RUNG DROP for both fleet-global values above (DESIGN 4b(3)).
// ============================================================================================
//
// WHY ONE ROW AND NOT TWO. The context ceiling and the slot count are the same defect in two
// spellings: a value measured on one node, one artifact, one runtime configuration and one
// concurrency condition, then applied to two hosts and every realization the cell can load. Two
// rows would be two places to update when one restoration discharges both, and the restoration IS
// shared -- neither value becomes qualified until per-realization observation exists.
//
// WHAT WAS ACTUALLY MEASURED, so the gap between evidence and scope is legible rather than implied:
// spark-a3ee, the hf.co/antirez/deepseek-v4-gguf IQ2_XXS build, Ollama 0.32.9, one resident session,
// buffer figures read from /api/ps. That is one cell of a matrix whose other axes are the six other
// resident models, the second host, and every concurrency above one.
//
// THE TRIGGER NAMES THE CAPABILITY AND NOT AN ARTIFACT. A per-realization observation table is not
// sufficient by itself: while a global default exists, an unmeasured realization still INHERITS
// these numbers automatically, so measuring some of the roster leaves the same unqualified
// inheritance for the rest. The capability is that no realization can be served at a configuration
// it has not been observed at -- which requires both the observations and the removal of automatic
// inheritance.
data spark_serving_fleet_global_configuration_drop: GuaranteeRungDrop = GuaranteeRungDrop {
subject: "the serving cell's default context window and concurrent slot count, applied fleet-wide from single-realization evidence",
previous: MechanicallyPreventable,
temporary: Mitigatable,
reason: DeletedWithoutReplacement,
population: [
"gunbc.spark.serving_desired spark_serving_desired_context_length",
"gunbc.spark.serving_desired spark_serving_desired_serving_slots_value",
],
restoration_trigger: "no realization can be served at a configuration it has not been observed at: a per-realization runner-memory observation exists for every model the cell loads, at the configuration selected for it, AND the fleet-global default no longer confers an unmeasured realization automatic admission -- an unobserved realization must refuse rather than inherit. An observation table alone does not discharge this row, because inheritance is the half that makes an unmeasured model's admission silent",
}
15 changes: 9 additions & 6 deletions dag/gunbc/spark/serving_install_paths.dag
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
module gunbc.spark.serving_install_paths

import std.types { String, FilePath, NonEmptyStr }
import std.measure { TokenCount }
import std.measure { TokenCount, PositiveSlotCount }
import std.content_hash { Fnv1a64Structural }
import gunbc.spark.serving_runtime_receipt { spark_serving_executor_home }
import gunbc.spark.serving_unit_identity { spark_serving_unit_name }
Expand Down Expand Up @@ -56,7 +56,8 @@ type SparkServingUserUnitSpec {
exec_subcommand: NonEmptyStr
restart: NonEmptyStr
wanted_by: NonEmptyStr
configured_context: TokenCount
default_context: TokenCount
serving_slots: PositiveSlotCount
}

// THE MODEL STORE PATHS, derived from the same models root the unit's OLLAMA_MODELS points at, so
Expand Down Expand Up @@ -311,14 +312,15 @@ fn spark_serving_user_unit_path() -> FilePath {
) as FilePath
}

// `configured_context` is a PARAMETER rather than a literal here for the same reason
// `default_context` is a PARAMETER rather than a literal here for the same reason
// `bind_host_port` is: it is desired state, and its authority is the occurrence's launch profile
// (gunbc.spark.serving_desired spark_serving_desired_configured_context). Spelling a number here
// (gunbc.spark.serving_desired spark_serving_desired_default_context). Spelling a number here
// would make this module a second authority for the window and would put the declared profile back
// out of the unit's reach -- exactly the defect this change closes.
fn spark_serving_user_unit_spec(
bind_host_port: NonEmptyStr,
configured_context: TokenCount,
default_context: TokenCount,
serving_slots: PositiveSlotCount,
) -> SparkServingUserUnitSpec {
SparkServingUserUnitSpec {
description: "gunbc Spark serving (Ollama gpt-oss:20b)" as NonEmptyStr,
Expand All @@ -332,7 +334,8 @@ fn spark_serving_user_unit_spec(
exec_subcommand: "serve" as NonEmptyStr,
restart: "on-failure" as NonEmptyStr,
wanted_by: "default.target" as NonEmptyStr,
configured_context: configured_context,
default_context: default_context,
serving_slots: serving_slots,
}
}

6 changes: 4 additions & 2 deletions dag/gunbc/spark/serving_membership.dag
Original file line number Diff line number Diff line change
Expand Up @@ -26,7 +26,8 @@ import gunbc.spark.serving_install_paths {
import gunbc.spark.serving_desired {
spark_serving_desired_bind_host_port,
spark_serving_automation_principal_login,
spark_serving_desired_configured_context,
spark_serving_desired_default_context,
spark_serving_desired_serving_slots,
}
import gunbc.spark.serving_unit_identity { spark_serving_unit_name }
import gunbc.spark.serving_unit_render { spark_serving_desired_user_unit_content_identity }
Expand Down Expand Up @@ -310,7 +311,8 @@ fn spark_serving_user_unit_file_write_intent() -> SparkServingUserUnitWriteInten
path: spark_serving_user_unit_path(),
spec: spark_serving_user_unit_spec(
bind_host_port: spark_serving_desired_bind_host_port() as NonEmptyStr,
configured_context: spark_serving_desired_configured_context(),
default_context: spark_serving_desired_default_context(),
serving_slots: spark_serving_desired_serving_slots(),
),
}
}
Expand Down
5 changes: 3 additions & 2 deletions dag/gunbc/spark/serving_release.dag
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@ import gunbc.model_artifact { ModelArtifact }
import extdeps.ollama.api { OllamaModelRef, ollama_model_ref_wire, ollama_default_port }
import v2.std.algebra { fold_list }
import std.nat { Nat }
import std.measure { TokenCount }
import std.measure { TokenCount, PositiveSlotCount }
import gunbc.spark.serving_realization {
SystemPrincipal, ServingDeploymentTarget, UserUnitRealization,
EscalationShortfall, NoNewPrivilegeRequired, NewPrivilegeRequired, ShortfallUndecidable,
Expand Down Expand Up @@ -83,7 +83,8 @@ type SparkServingRelease {
// identity domains.
type OllamaServingLaunchProfile {
bind: SparkServingBindListen
configured_context: TokenCount
default_context: TokenCount
serving_slots: PositiveSlotCount
}

// The identity is DERIVED from the release, never authored. It previously carried a NonEmptyStr
Expand Down
Loading
Loading