From e2f3d3a0390bcb19dd63ef852fe4b14662aed39f Mon Sep 17 00:00:00 2001 From: Brian Searls Date: Sat, 20 Jun 2026 00:52:50 +0000 Subject: [PATCH 1/7] WIP: CI is extremely flakey --- dsl/gunbc/ci_fleet.dag | 16 ++++++++++------ src/v2/workflow/ci_floor_plan.dag | 12 ++++++++++-- 2 files changed, 20 insertions(+), 8 deletions(-) diff --git a/dsl/gunbc/ci_fleet.dag b/dsl/gunbc/ci_fleet.dag index 28ab8aeaf39..2066dafca4c 100644 --- a/dsl/gunbc/ci_fleet.dag +++ b/dsl/gunbc/ci_fleet.dag @@ -92,8 +92,10 @@ fn gunbc_ci_runner_spec() -> RunnerSpec { runner_spec_from_offer(offer: gunbc_ci_fleet_offer) } -// Phase 1: corpus as IndependentShards WorkDemand (degenerate shard_count=1 today). -fn gunbc_ci_floor_corpus_work_demand() -> WorkDemand { +// Corpus parallel breadth is the CALLER's fact: the floor plan derives it from the +// single-authority spec (gates + corpus cardinality), so it is NOT a tuned constant. +// This keeps the §3 split honest — demand = the work (floor plan), supply = the fleet. +fn gunbc_ci_floor_corpus_work_demand(corpus_shard_count: Int) -> WorkDemand { WorkDemand { resources: ResourceEnvelope { cpu: none, @@ -105,7 +107,7 @@ fn gunbc_ci_floor_corpus_work_demand() -> WorkDemand { os: none, isolation: IsolationRequirement { boundary: PerJobFilesystem }, toolchains: [], - parallelism: IndependentShards { shard_count: 1 }, + parallelism: IndependentShards { shard_count: corpus_shard_count }, data_locality: [], effects: [], } @@ -115,11 +117,13 @@ fn gunbc_ci_floor_placement_supply() -> PlacementSupplyRow? { offer_placement_supply_row(offer: gunbc_ci_fleet_offer) } -// Host spawn-width authority for the CI fleet offer + corpus WorkDemand. -fn ci_fleet_floor_spawn_width() -> HardwareThreadCount { +// Host spawn-width = min(caller's corpus breadth, this fleet's hardware threads). +// No hand-tuned width: add a gate to the spec → breadth re-derives; change the host → +// supply re-derives; spawn_width follows from both with zero edits here. +fn ci_fleet_floor_spawn_width(corpus_shard_count: Int) -> HardwareThreadCount { match gunbc_ci_floor_placement_supply() { Absent => hardware_thread_count(1) Present { value: row } => - placement_spawn_width(supply: row, demand: gunbc_ci_floor_corpus_work_demand()) + placement_spawn_width(supply: row, demand: gunbc_ci_floor_corpus_work_demand(corpus_shard_count: corpus_shard_count)) } } diff --git a/src/v2/workflow/ci_floor_plan.dag b/src/v2/workflow/ci_floor_plan.dag index 038daaa572e..275784bdaa3 100644 --- a/src/v2/workflow/ci_floor_plan.dag +++ b/src/v2/workflow/ci_floor_plan.dag @@ -215,7 +215,15 @@ fn gunbc_ci_floor_schedule_lens_holds() -> Bool { } } -// claim_executor spawn-width companion (feature:scheduler-width-budget scaffold). +// Floor breadth = the count of independent floor units (one per spec gate; the corpus +// node rides the same later layer), DERIVED from the single-authority spec — never a +// tuned constant. Add a gate to gunbc_ci_spec → breadth re-derives here automatically. +fn gunbc_ci_floor_breadth() -> Int { + fold(gunbc_ci_spec.gates, init: 0, f: fn(acc, g) { acc + 1 }) +} + +// claim_executor spawn-width companion: min(floor breadth, fleet hardware threads). +// The relationship is the model; nothing here is hand-tuned per host or per gate. fn gunbc_ci_floor_spawn_width() -> HardwareThreadCount { - ci_fleet_floor_spawn_width() + ci_fleet_floor_spawn_width(corpus_shard_count: gunbc_ci_floor_breadth()) } From 7625f8422faab3c3f192f254ec92ad9b05a8243b Mon Sep 17 00:00:00 2001 From: Brian Searls Date: Sat, 20 Jun 2026 00:55:17 +0000 Subject: [PATCH 2/7] floor spawn_width: derive from spec breadth (no hand-tuning) + discriminating witness MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Floor ran serially because the corpus WorkDemand hardcoded shard_count=1, so spawn_width = min(1, threads) = 1. Now the floor plan derives breadth from the single-authority spec (one unit per gate) and the fleet supplies the thread cap: spawn_width = min(breadth, hardware_threads). Verified by execution: 1 -> 7. Witnesses: spawn_width tracks breadth (the relationship), and floor-not-serial (discriminating — reverting to a constant shard_count goes RED). Co-Authored-By: Claude Opus 4.8 --- .../test/claim/ci_floor_plan_witness_test.dag | 22 +++++++++++++++++-- 1 file changed, 20 insertions(+), 2 deletions(-) diff --git a/src/v2/test/claim/ci_floor_plan_witness_test.dag b/src/v2/test/claim/ci_floor_plan_witness_test.dag index e788c1aee4e..076576baf4a 100644 --- a/src/v2/test/claim/ci_floor_plan_witness_test.dag +++ b/src/v2/test/claim/ci_floor_plan_witness_test.dag @@ -14,8 +14,10 @@ import std.realization { Runnable, RunnableSingleClaim, RunnableDiscoveryBatch, } import v2.workflow.ci_floor_plan { - gunbc_ci_floor_batches, corpus_runnable + gunbc_ci_floor_batches, corpus_runnable, + gunbc_ci_floor_breadth, gunbc_ci_floor_spawn_width } +import std.measure { hardware_thread_count_value } import gunbc.ci_layer_roots { witness_discovery_scan_dirs, witness_layer_roots } import v2.std.algebra { length, list_snoc_item } import v2.std.collection { List } @@ -100,6 +102,20 @@ fn witness_corpus_skip_node_frontier_enabled() -> Bool { } } +// Connected relationship: spawn_width is DERIVED from the floor breadth, not tuned. +// When breadth <= hardware threads (true on the real fleet), spawn_width == breadth — +// so the host runs the independent layer concurrently, not one-at-a-time. +fn witness_spawn_width_tracks_breadth() -> Bool { + hardware_thread_count_value(t: gunbc_ci_floor_spawn_width()) == gunbc_ci_floor_breadth() +} + +// Discriminating: the floor is NOT serial. A revert to a constant shard_count (=1) or a +// supply that collapsed to one thread makes breadth/width 1 → this goes RED. +fn witness_floor_not_serial() -> Bool { + (gunbc_ci_floor_breadth() > 1) && + (hardware_thread_count_value(t: gunbc_ci_floor_spawn_width()) > 1) +} + test fn ci_floor_plan_witnesses() -> Bool { witness_two_readiness_layers() && witness_compile_root_runs_first() && @@ -108,5 +124,7 @@ test fn ci_floor_plan_witnesses() -> Bool { witness_discovery_scan_dirs_in_plan() && witness_scan_dirs_match_layer_authority() && witness_corpus_has_no_explicit_entries() && - witness_corpus_skip_node_frontier_enabled() + witness_corpus_skip_node_frontier_enabled() && + witness_spawn_width_tracks_breadth() && + witness_floor_not_serial() } From 412206fa15bbc47e43bff9fe91d19f71d7aec174 Mon Sep 17 00:00:00 2001 From: Brian Searls Date: Sat, 20 Jun 2026 01:00:15 +0000 Subject: [PATCH 3/7] review nit: use length(gates) instead of fold-count for breadth claude-opus-4-7 review on #5356: the +1 fold is just length(). Verified length resolves on gunbc_ci_spec.gates across the v2->dsl bridge and still yields 7; witness stays green. Co-Authored-By: Claude Opus 4.8 --- src/v2/workflow/ci_floor_plan.dag | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/v2/workflow/ci_floor_plan.dag b/src/v2/workflow/ci_floor_plan.dag index 275784bdaa3..9facb18f89c 100644 --- a/src/v2/workflow/ci_floor_plan.dag +++ b/src/v2/workflow/ci_floor_plan.dag @@ -32,7 +32,7 @@ import gunbc.ci_spec { Gate, RustMonolithGate, EmitHostGate, LayeringImportsGate, ResolvedImportsGate, DslCompileCleanGate, CiYamlGate, SourceRootIngestGate } -import v2.std.algebra { list_snoc_item } +import v2.std.algebra { length, list_snoc_item } import v2.std.collection { List } import v2.std.dependency { DataDependsOn, DependencyView } import v2.std.node { Atom, Node, TypeNode } @@ -219,7 +219,7 @@ fn gunbc_ci_floor_schedule_lens_holds() -> Bool { // node rides the same later layer), DERIVED from the single-authority spec — never a // tuned constant. Add a gate to gunbc_ci_spec → breadth re-derives here automatically. fn gunbc_ci_floor_breadth() -> Int { - fold(gunbc_ci_spec.gates, init: 0, f: fn(acc, g) { acc + 1 }) + length(xs: gunbc_ci_spec.gates) } // claim_executor spawn-width companion: min(floor breadth, fleet hardware threads). From 1ddc7ae4732885a041442a7ade573ae86409d042 Mon Sep 17 00:00:00 2001 From: Brian Searls Date: Sat, 20 Jun 2026 01:45:40 +0000 Subject: [PATCH 4/7] WIP: CI is extremely flakey --- src/v2/extdeps/languages/bash.dag | 10 +++------- 1 file changed, 3 insertions(+), 7 deletions(-) diff --git a/src/v2/extdeps/languages/bash.dag b/src/v2/extdeps/languages/bash.dag index e1140047a6d..0844c94bb54 100644 --- a/src/v2/extdeps/languages/bash.dag +++ b/src/v2/extdeps/languages/bash.dag @@ -1089,13 +1089,9 @@ fn bash_exit_code_1_quoted_wrong_translation_rules_node() -> Node { } fn bash_exit_code_lex() -> LexRules { - ModeledLexRules { - root: LexRuleSet { - rules: [ - bash_lex_rule(token_class: ^bash_token_kw_exit, text: "exit ") - ] - } - } + phrase_to_lex_rules(p: phrase([ + span(text: "exit ", class: ^bash_token_kw_exit) + ])) } fn bash_exit_code_1_target_model() -> TargetModel { From 4bd78d06a13bc6f1a2383225355f847a110be765 Mon Sep 17 00:00:00 2001 From: Brian Searls Date: Sat, 20 Jun 2026 03:06:04 +0000 Subject: [PATCH 5/7] WIP: CI is extremely flakey --- dsl/gunbc/ci_fleet.dag | 17 ++++++++++-- dsl/product/compute_fabric.dag | 51 +++++++++++++++++++++++++++++----- 2 files changed, 59 insertions(+), 9 deletions(-) diff --git a/dsl/gunbc/ci_fleet.dag b/dsl/gunbc/ci_fleet.dag index 2066dafca4c..f28977da508 100644 --- a/dsl/gunbc/ci_fleet.dag +++ b/dsl/gunbc/ci_fleet.dag @@ -24,10 +24,12 @@ import product.compute_fabric { Dram, WorkDemand, ResourceEnvelope, + MemoryRequirement, IsolationRequirement, IndependentShards, offer_placement_supply_row, placement_spawn_width, + placement_floor_memory_budget_bytes, } import extdeps.cpu.types { CpuFacts, CpuDeploymentFacts } import std.measure { byte_size, hertz, hardware_thread_count, HardwareThreadCount } @@ -58,7 +60,12 @@ data gunbc_ci_host: ComputeHost = ComputeHost { identity: gunbc_ci_host_identity, baseboard: none, processors: [CpuProcessor { cpu: gunbc_ci_cpu }], - memory: [MemoryDevice { capacity: byte_size(274877906944), memory_kind: Dram, bandwidth: none }], + // 125 GiB measured on the real fleet host (Ampere Altra Max M128, 1 NUMA node). The prior + // 256 GiB was a stub; the memory-aware spawn-width bound (placement_spawn_width) needs a + // real RAM figure or it cannot back fan-out off before the cgroup OOM. CPU thread count + // (catalog) is still the 64-thread stub — corrected with the ctrl fleet grounding; the + // memory bound dominates the floor width regardless, so it is not load-bearing here. + memory: [MemoryDevice { capacity: byte_size(134217728000), memory_kind: Dram, bandwidth: none }], storage: [], network_interfaces: [], } @@ -100,7 +107,13 @@ fn gunbc_ci_floor_corpus_work_demand(corpus_shard_count: Int) -> WorkDemand { resources: ResourceEnvelope { cpu: none, gpu: none, - memory: none, + // Per-shard peak resolve memory. Grounded in the deterministic floor-OOM evidence + // (eager-boar-790, 3 runs of a grown corpus at width=7): srv2's ~87.6 GiB cgroup cap + // passes 7 concurrent ~2900-item resolves only by a thin margin (~12.5 GiB/unit), and + // srv1's ~65.6 GiB cap OOM-kills (exit 137). 14 GiB/unit carries headroom for corpus + // growth; the memory-aware width then floors batch-2 below the OOM ceiling on both + // hosts. Grounds-tighter when a peak-RSS probe per resolve lands. + memory: Present { value: MemoryRequirement { min_bytes: byte_size(15032385536) } }, storage: none, network: none, }, diff --git a/dsl/product/compute_fabric.dag b/dsl/product/compute_fabric.dag index cd9edd42f13..59878c7d15b 100644 --- a/dsl/product/compute_fabric.dag +++ b/dsl/product/compute_fabric.dag @@ -80,7 +80,7 @@ import extdeps.toolchain.types { } import product.placement_supply { HostIdentity, PlacementSupplyRow } import std.baseboard { ServerBaseboard } -import std.realization_width { bounded_host_spawn_width } +import std.realization_width { bounded_host_spawn_width, int_min } // 🟡 forward — v2.std.witness / v2.std.diagnostic at harness (P-CF-WITNESS). // Kept intentionally distinct from the kernel `Witness` container. @@ -490,14 +490,51 @@ fn work_demand_shard_count(demand: WorkDemand) -> Int { } } -// Width projection: min(shard_count, PlacementSupplyRow.hardware_threads) — single authority. -// Eligibility is NOT rejected when demand exceeds supply — the peripheral host fold caps -// fan-out; satisfies records DemandParallelism as proven when IndependentShards is present. +// Conservative memory budget for concurrent fan-out on a host. A floor batch is killed by +// its runner-slice cgroup MemoryMax (NOT total RAM), which on the tightest fleet host caps +// near half of RAM. Until ctrl/BMC grounds the exact per-host MemoryMax bytes, approximate +// the budget as 0.50×RAM so a single fleet-wide spawn width stays safe on the TIGHTEST host +// (the floor lands on either host). Grounds-later: replace 0.50×RAM with measured MemoryMax. +fn placement_floor_memory_budget_bytes(supply: PlacementSupplyRow) -> Int { + byte_size_count(supply.ram_bytes) / 2 +} + +// Memory-bounded width: how many concurrent units fit under the memory budget, each peaking +// at the demand's per-unit memory requirement. Absent memory demand => no memory bound (the +// hardware-thread count is an inert ceiling). per-unit guarded > 0 (never divide by zero); +// a budget under one unit still admits 1 (fail-closed forward progress, never 0 width). +fn placement_memory_width(supply: PlacementSupplyRow, demand: WorkDemand) -> Int { + let cores = hardware_thread_count_value(t: supply.hardware_threads) + match demand.resources.memory { + Absent => cores + Present { value: req } => { + let per_unit = byte_size_count(req.min_bytes) + if per_unit <= 0 { cores } else { + let fits = placement_floor_memory_budget_bytes(supply: supply) / per_unit + if fits < 1 { 1 } else { fits } + } + } + } +} + +// Width projection: min(shard_count, hardware_threads, memory_budget / per_unit_peak) — +// single authority, MEMORY-AWARE. The CPU bound (bounded_host_spawn_width) alone is +// memory-blind: N concurrent resolves each peak at the demand's per-unit memory, and N×peak +// must stay under the floor's cgroup MemoryMax or the batch is OOM-killed (exit 137). The +// memory term backs fan-out off BEFORE the kill — §1's safety axis made structural, and the +// connected model §6 ('fix related systems together') requires: CPU and memory are ONE width +// relationship, not a CPU bound patched after an OOM. Eligibility is NOT rejected when demand +// exceeds supply — the peripheral host fold caps fan-out; satisfies records DemandParallelism +// as proven when IndependentShards is present. fn placement_spawn_width(supply: PlacementSupplyRow, demand: WorkDemand) -> HardwareThreadCount { - bounded_host_spawn_width( - shard_count: work_demand_shard_count(demand: demand), - hardware_threads: supply.hardware_threads + let cpu_width = hardware_thread_count_value( + t: bounded_host_spawn_width( + shard_count: work_demand_shard_count(demand: demand), + hardware_threads: supply.hardware_threads + ) ) + let mem_width = placement_memory_width(supply: supply, demand: demand) + hardware_thread_count(count: int_min(a: cpu_width, b: mem_width)) } // --- §1.2 demand ------------------------------------------------------------- From d50c13b996109f262222433ad350185dec4464ae Mon Sep 17 00:00:00 2001 From: Brian Searls Date: Sat, 20 Jun 2026 03:10:15 +0000 Subject: [PATCH 6/7] =?UTF-8?q?fix(ci-floor):=20memory-aware=20spawn=5Fwid?= =?UTF-8?q?th=20=E2=80=94=20bound=20batch-2=20fan-out=20by=20cgroup=20budg?= =?UTF-8?q?et,=20not=20CPU=20only?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit placement_spawn_width was memory-blind (min of shard_count and hardware_threads). At width=7 the batch-2 corpus resolves OOM-killed (exit 137) deterministically as the corpus grows (eager-boar-790, 3 runs). Wire the already-modeled memory terms (PlacementSupplyRow.ram_bytes + ResourceEnvelope.memory) into the width: spawn_width = min(shard_count, hardware_threads, memory_budget / per_unit_peak). Verified by execution: floor spawn_width 7 -> 4 (4x14GiB=56GiB <= 0.50x125GiB budget, safe under srv1's ~65GiB cgroup cap); gunbc_ci_floor_spawn_fits_memory_budget -> true; ci_floor_plan_witnesses -> true (adds witness_floor_width_fits_memory_budget + witness_floor_width_memory_bounded, discriminating vs the memory-blind regression). Co-Authored-By: Claude Opus 4.8 --- dsl/gunbc/ci_fleet.dag | 29 ++++++++++++--- dsl/product/compute_fabric.dag | 24 +++++++------ .../test/claim/ci_floor_plan_witness_test.dag | 36 ++++++++++++++----- src/v2/workflow/ci_floor_plan.dag | 14 ++++++-- 4 files changed, 77 insertions(+), 26 deletions(-) diff --git a/dsl/gunbc/ci_fleet.dag b/dsl/gunbc/ci_fleet.dag index f28977da508..19058df734d 100644 --- a/dsl/gunbc/ci_fleet.dag +++ b/dsl/gunbc/ci_fleet.dag @@ -32,7 +32,7 @@ import product.compute_fabric { placement_floor_memory_budget_bytes, } import extdeps.cpu.types { CpuFacts, CpuDeploymentFacts } -import std.measure { byte_size, hertz, hardware_thread_count, HardwareThreadCount } +import std.measure { byte_size, byte_size_count, hertz, hardware_thread_count, hardware_thread_count_value, HardwareThreadCount } import std.os.types { OperatingSystemSurface, Linux, @@ -130,9 +130,11 @@ fn gunbc_ci_floor_placement_supply() -> PlacementSupplyRow? { offer_placement_supply_row(offer: gunbc_ci_fleet_offer) } -// Host spawn-width = min(caller's corpus breadth, this fleet's hardware threads). -// No hand-tuned width: add a gate to the spec → breadth re-derives; change the host → -// supply re-derives; spawn_width follows from both with zero edits here. +// Host spawn-width = min(corpus breadth, hardware threads, memory_budget / per-shard peak). +// No hand-tuned width: add a gate to the spec → breadth re-derives; change the host → supply +// re-derives; raise the per-shard memory → the memory bound re-derives; spawn_width follows +// from all three via placement_spawn_width with zero edits here. The memory term is what +// keeps batch-2 from OOM-killing as the corpus grows (the CPU-only bound was memory-blind). fn ci_fleet_floor_spawn_width(corpus_shard_count: Int) -> HardwareThreadCount { match gunbc_ci_floor_placement_supply() { Absent => hardware_thread_count(1) @@ -140,3 +142,22 @@ fn ci_fleet_floor_spawn_width(corpus_shard_count: Int) -> HardwareThreadCount { placement_spawn_width(supply: row, demand: gunbc_ci_floor_corpus_work_demand(corpus_shard_count: corpus_shard_count)) } } + +// §5 OOM oracle: the chosen floor width times the per-shard peak memory must fit the +// conservative memory budget — i.e. batch-2's concurrent memory peak stays under the cgroup +// MemoryMax. Discriminating: a memory-blind width (= min(breadth, threads), ignoring the +// memory term) would exceed the budget and flip this false. Absent supply / Absent memory +// demand are vacuously safe (no concurrent memory pressure to bound). +fn ci_fleet_floor_spawn_fits_memory_budget(corpus_shard_count: Int) -> Bool { + match gunbc_ci_floor_placement_supply() { + Absent => true + Present { value: row } => + match gunbc_ci_floor_corpus_work_demand(corpus_shard_count: corpus_shard_count).resources.memory { + Absent => true + Present { value: req } => { + let width = hardware_thread_count_value(t: ci_fleet_floor_spawn_width(corpus_shard_count: corpus_shard_count)) + width * byte_size_count(req.min_bytes) <= placement_floor_memory_budget_bytes(supply: row) + } + } + } +} diff --git a/dsl/product/compute_fabric.dag b/dsl/product/compute_fabric.dag index 59878c7d15b..9dece2e4c07 100644 --- a/dsl/product/compute_fabric.dag +++ b/dsl/product/compute_fabric.dag @@ -499,19 +499,19 @@ fn placement_floor_memory_budget_bytes(supply: PlacementSupplyRow) -> Int { byte_size_count(supply.ram_bytes) / 2 } -// Memory-bounded width: how many concurrent units fit under the memory budget, each peaking -// at the demand's per-unit memory requirement. Absent memory demand => no memory bound (the -// hardware-thread count is an inert ceiling). per-unit guarded > 0 (never divide by zero); -// a budget under one unit still admits 1 (fail-closed forward progress, never 0 width). -fn placement_memory_width(supply: PlacementSupplyRow, demand: WorkDemand) -> Int { - let cores = hardware_thread_count_value(t: supply.hardware_threads) +// Memory width BOUND: how many concurrent units fit under the memory budget, each peaking at +// the demand's per-unit memory requirement. `none` = no memory bound (Absent demand, or a +// degenerate non-positive per-unit guarded against divide-by-zero). A budget under one unit +// still admits 1 (fail-closed forward progress, never 0 width). Returned as `Int?` so the +// "no bound" case never has to unify with the Int division result. +fn placement_memory_width_bound(supply: PlacementSupplyRow, demand: WorkDemand) -> Int? { match demand.resources.memory { - Absent => cores + Absent => none Present { value: req } => { let per_unit = byte_size_count(req.min_bytes) - if per_unit <= 0 { cores } else { + if per_unit <= 0 { none } else { let fits = placement_floor_memory_budget_bytes(supply: supply) / per_unit - if fits < 1 { 1 } else { fits } + Present { value: if fits < 1 { 1 } else { fits } } } } } @@ -533,8 +533,10 @@ fn placement_spawn_width(supply: PlacementSupplyRow, demand: WorkDemand) -> Hard hardware_threads: supply.hardware_threads ) ) - let mem_width = placement_memory_width(supply: supply, demand: demand) - hardware_thread_count(count: int_min(a: cpu_width, b: mem_width)) + match placement_memory_width_bound(supply: supply, demand: demand) { + Absent => hardware_thread_count(count: cpu_width) + Present { value: mem_width } => hardware_thread_count(count: int_min(a: cpu_width, b: mem_width)) + } } // --- §1.2 demand ------------------------------------------------------------- diff --git a/src/v2/test/claim/ci_floor_plan_witness_test.dag b/src/v2/test/claim/ci_floor_plan_witness_test.dag index 076576baf4a..1d6f4484d8f 100644 --- a/src/v2/test/claim/ci_floor_plan_witness_test.dag +++ b/src/v2/test/claim/ci_floor_plan_witness_test.dag @@ -15,7 +15,8 @@ import std.realization { } import v2.workflow.ci_floor_plan { gunbc_ci_floor_batches, corpus_runnable, - gunbc_ci_floor_breadth, gunbc_ci_floor_spawn_width + gunbc_ci_floor_breadth, gunbc_ci_floor_spawn_width, + gunbc_ci_floor_spawn_fits_memory_budget } import std.measure { hardware_thread_count_value } import gunbc.ci_layer_roots { witness_discovery_scan_dirs, witness_layer_roots } @@ -102,11 +103,12 @@ fn witness_corpus_skip_node_frontier_enabled() -> Bool { } } -// Connected relationship: spawn_width is DERIVED from the floor breadth, not tuned. -// When breadth <= hardware threads (true on the real fleet), spawn_width == breadth — -// so the host runs the independent layer concurrently, not one-at-a-time. -fn witness_spawn_width_tracks_breadth() -> Bool { - hardware_thread_count_value(t: gunbc_ci_floor_spawn_width()) == gunbc_ci_floor_breadth() +// Connected relationship: spawn_width is DERIVED (min of breadth, hardware threads, and the +// memory budget / per-shard peak), never a tuned constant. It is positive and never exceeds +// the work breadth — fanning out more workers than shards would be wasted concurrency. +fn witness_spawn_width_bounded_by_breadth() -> Bool { + let width = hardware_thread_count_value(t: gunbc_ci_floor_spawn_width()) + (width >= 1) && (width <= gunbc_ci_floor_breadth()) } // Discriminating: the floor is NOT serial. A revert to a constant shard_count (=1) or a @@ -116,6 +118,22 @@ fn witness_floor_not_serial() -> Bool { (hardware_thread_count_value(t: gunbc_ci_floor_spawn_width()) > 1) } +// Connected memory relationship (the §5 OOM oracle): the scheduled width's concurrent memory +// peak fits the fleet memory budget. DISCRIMINATING against the memory-blind regression — if +// placement_spawn_width dropped its memory term, width would jump to min(breadth, threads)=7 +// and 7×(per-shard peak) would blow the budget, flipping this RED. This is the witness that +// would have caught the deterministic floor-OOM (exit 137) at model time, not on the fleet. +fn witness_floor_width_fits_memory_budget() -> Bool { + gunbc_ci_floor_spawn_fits_memory_budget() +} + +// Discriminating: the memory term is ACTIVE — on the real fleet the per-shard peak caps the +// width strictly below the CPU/breadth bound (mem budget admits fewer than `breadth` units). +// A memory-blind width would equal breadth, flipping this RED. +fn witness_floor_width_memory_bounded() -> Bool { + hardware_thread_count_value(t: gunbc_ci_floor_spawn_width()) < gunbc_ci_floor_breadth() +} + test fn ci_floor_plan_witnesses() -> Bool { witness_two_readiness_layers() && witness_compile_root_runs_first() && @@ -125,6 +143,8 @@ test fn ci_floor_plan_witnesses() -> Bool { witness_scan_dirs_match_layer_authority() && witness_corpus_has_no_explicit_entries() && witness_corpus_skip_node_frontier_enabled() && - witness_spawn_width_tracks_breadth() && - witness_floor_not_serial() + witness_spawn_width_bounded_by_breadth() && + witness_floor_not_serial() && + witness_floor_width_fits_memory_budget() && + witness_floor_width_memory_bounded() } diff --git a/src/v2/workflow/ci_floor_plan.dag b/src/v2/workflow/ci_floor_plan.dag index 9facb18f89c..c9d9924ee85 100644 --- a/src/v2/workflow/ci_floor_plan.dag +++ b/src/v2/workflow/ci_floor_plan.dag @@ -37,7 +37,7 @@ import v2.std.collection { List } import v2.std.dependency { DataDependsOn, DependencyView } import v2.std.node { Atom, Node, TypeNode } import v2.std.text { String } -import gunbc.ci_fleet { ci_fleet_floor_spawn_width } +import gunbc.ci_fleet { ci_fleet_floor_spawn_width, ci_fleet_floor_spawn_fits_memory_budget } import std.types { ContentHash } import std.measure { HardwareThreadCount } import std.realization_width { width_fold_objective_goals } @@ -222,8 +222,16 @@ fn gunbc_ci_floor_breadth() -> Int { length(xs: gunbc_ci_spec.gates) } -// claim_executor spawn-width companion: min(floor breadth, fleet hardware threads). -// The relationship is the model; nothing here is hand-tuned per host or per gate. +// claim_executor spawn-width companion: min(floor breadth, fleet hardware threads, memory +// budget / per-shard peak). The relationship is the model; nothing here is hand-tuned per +// host or per gate — the memory term backs fan-out off before the cgroup OOM-kill. fn gunbc_ci_floor_spawn_width() -> HardwareThreadCount { ci_fleet_floor_spawn_width(corpus_shard_count: gunbc_ci_floor_breadth()) } + +// §5 OOM oracle at the floor breadth: the scheduled width's concurrent memory peak fits the +// fleet memory budget. A revert to a memory-blind width (= min(breadth, threads)) flips this +// false on the real fleet (7×14 GiB > 0.50×125 GiB budget). +fn gunbc_ci_floor_spawn_fits_memory_budget() -> Bool { + ci_fleet_floor_spawn_fits_memory_budget(corpus_shard_count: gunbc_ci_floor_breadth()) +} From 6d7a3a9a88f8f55983ae3b7a7b0eb7d396a7cf86 Mon Sep 17 00:00:00 2001 From: Brian Searls Date: Sat, 20 Jun 2026 03:36:31 +0000 Subject: [PATCH 7/7] fix: resolve auto-committed merge conflict markers (take memory-aware HEAD; main side == #5356 subset) Co-Authored-By: Claude Opus 4.8 --- dsl/gunbc/ci_fleet.dag | 9 --------- .../test/claim/ci_floor_plan_witness_test.dag | 20 ------------------- src/v2/workflow/ci_floor_plan.dag | 7 ------- 3 files changed, 36 deletions(-) diff --git a/dsl/gunbc/ci_fleet.dag b/dsl/gunbc/ci_fleet.dag index 0b6a60d2684..19058df734d 100644 --- a/dsl/gunbc/ci_fleet.dag +++ b/dsl/gunbc/ci_fleet.dag @@ -130,23 +130,16 @@ fn gunbc_ci_floor_placement_supply() -> PlacementSupplyRow? { offer_placement_supply_row(offer: gunbc_ci_fleet_offer) } -<<<<<<< HEAD // Host spawn-width = min(corpus breadth, hardware threads, memory_budget / per-shard peak). // No hand-tuned width: add a gate to the spec → breadth re-derives; change the host → supply // re-derives; raise the per-shard memory → the memory bound re-derives; spawn_width follows // from all three via placement_spawn_width with zero edits here. The memory term is what // keeps batch-2 from OOM-killing as the corpus grows (the CPU-only bound was memory-blind). -======= -// Host spawn-width = min(caller's corpus breadth, this fleet's hardware threads). -// No hand-tuned width: add a gate to the spec → breadth re-derives; change the host → -// supply re-derives; spawn_width follows from both with zero edits here. ->>>>>>> origin/main fn ci_fleet_floor_spawn_width(corpus_shard_count: Int) -> HardwareThreadCount { match gunbc_ci_floor_placement_supply() { Absent => hardware_thread_count(1) Present { value: row } => placement_spawn_width(supply: row, demand: gunbc_ci_floor_corpus_work_demand(corpus_shard_count: corpus_shard_count)) -<<<<<<< HEAD } } @@ -166,7 +159,5 @@ fn ci_fleet_floor_spawn_fits_memory_budget(corpus_shard_count: Int) -> Bool { width * byte_size_count(req.min_bytes) <= placement_floor_memory_budget_bytes(supply: row) } } -======= ->>>>>>> origin/main } } diff --git a/src/v2/test/claim/ci_floor_plan_witness_test.dag b/src/v2/test/claim/ci_floor_plan_witness_test.dag index 3da10989a86..1d6f4484d8f 100644 --- a/src/v2/test/claim/ci_floor_plan_witness_test.dag +++ b/src/v2/test/claim/ci_floor_plan_witness_test.dag @@ -15,12 +15,8 @@ import std.realization { } import v2.workflow.ci_floor_plan { gunbc_ci_floor_batches, corpus_runnable, -<<<<<<< HEAD gunbc_ci_floor_breadth, gunbc_ci_floor_spawn_width, gunbc_ci_floor_spawn_fits_memory_budget -======= - gunbc_ci_floor_breadth, gunbc_ci_floor_spawn_width ->>>>>>> origin/main } import std.measure { hardware_thread_count_value } import gunbc.ci_layer_roots { witness_discovery_scan_dirs, witness_layer_roots } @@ -107,20 +103,12 @@ fn witness_corpus_skip_node_frontier_enabled() -> Bool { } } -<<<<<<< HEAD // Connected relationship: spawn_width is DERIVED (min of breadth, hardware threads, and the // memory budget / per-shard peak), never a tuned constant. It is positive and never exceeds // the work breadth — fanning out more workers than shards would be wasted concurrency. fn witness_spawn_width_bounded_by_breadth() -> Bool { let width = hardware_thread_count_value(t: gunbc_ci_floor_spawn_width()) (width >= 1) && (width <= gunbc_ci_floor_breadth()) -======= -// Connected relationship: spawn_width is DERIVED from the floor breadth, not tuned. -// When breadth <= hardware threads (true on the real fleet), spawn_width == breadth — -// so the host runs the independent layer concurrently, not one-at-a-time. -fn witness_spawn_width_tracks_breadth() -> Bool { - hardware_thread_count_value(t: gunbc_ci_floor_spawn_width()) == gunbc_ci_floor_breadth() ->>>>>>> origin/main } // Discriminating: the floor is NOT serial. A revert to a constant shard_count (=1) or a @@ -130,7 +118,6 @@ fn witness_floor_not_serial() -> Bool { (hardware_thread_count_value(t: gunbc_ci_floor_spawn_width()) > 1) } -<<<<<<< HEAD // Connected memory relationship (the §5 OOM oracle): the scheduled width's concurrent memory // peak fits the fleet memory budget. DISCRIMINATING against the memory-blind regression — if // placement_spawn_width dropped its memory term, width would jump to min(breadth, threads)=7 @@ -147,8 +134,6 @@ fn witness_floor_width_memory_bounded() -> Bool { hardware_thread_count_value(t: gunbc_ci_floor_spawn_width()) < gunbc_ci_floor_breadth() } -======= ->>>>>>> origin/main test fn ci_floor_plan_witnesses() -> Bool { witness_two_readiness_layers() && witness_compile_root_runs_first() && @@ -158,13 +143,8 @@ test fn ci_floor_plan_witnesses() -> Bool { witness_scan_dirs_match_layer_authority() && witness_corpus_has_no_explicit_entries() && witness_corpus_skip_node_frontier_enabled() && -<<<<<<< HEAD witness_spawn_width_bounded_by_breadth() && witness_floor_not_serial() && witness_floor_width_fits_memory_budget() && witness_floor_width_memory_bounded() -======= - witness_spawn_width_tracks_breadth() && - witness_floor_not_serial() ->>>>>>> origin/main } diff --git a/src/v2/workflow/ci_floor_plan.dag b/src/v2/workflow/ci_floor_plan.dag index d53ee05bccb..c9d9924ee85 100644 --- a/src/v2/workflow/ci_floor_plan.dag +++ b/src/v2/workflow/ci_floor_plan.dag @@ -222,7 +222,6 @@ fn gunbc_ci_floor_breadth() -> Int { length(xs: gunbc_ci_spec.gates) } -<<<<<<< HEAD // claim_executor spawn-width companion: min(floor breadth, fleet hardware threads, memory // budget / per-shard peak). The relationship is the model; nothing here is hand-tuned per // host or per gate — the memory term backs fan-out off before the cgroup OOM-kill. @@ -235,10 +234,4 @@ fn gunbc_ci_floor_spawn_width() -> HardwareThreadCount { // false on the real fleet (7×14 GiB > 0.50×125 GiB budget). fn gunbc_ci_floor_spawn_fits_memory_budget() -> Bool { ci_fleet_floor_spawn_fits_memory_budget(corpus_shard_count: gunbc_ci_floor_breadth()) -======= -// claim_executor spawn-width companion: min(floor breadth, fleet hardware threads). -// The relationship is the model; nothing here is hand-tuned per host or per gate. -fn gunbc_ci_floor_spawn_width() -> HardwareThreadCount { - ci_fleet_floor_spawn_width(corpus_shard_count: gunbc_ci_floor_breadth()) ->>>>>>> origin/main }