Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 3 additions & 3 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -111,7 +111,7 @@ jobs:
ci:
runs-on: [self-hosted, linux, arm64]
needs: [build]
timeout-minutes: 600
timeout-minutes: 130
steps:
- name: Checkout
uses: actions/checkout@v5
Expand Down Expand Up @@ -149,7 +149,7 @@ jobs:
fi
fi
"$ROOT/target/release/claim_executor" --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --plan-entry src/v2/workflow/ci_floor_plan.dag --plan-function gunbc_ci_regen_floor_batches --notice-title "self-host fixed-point (regen + staleness) — required; folded into ci job"
timeout-minutes: 270
timeout-minutes: 15
- name: Floor cgroup peak pre-read (calibration; reset if permitted)
run: |
d="/sys/fs/cgroup$(awk -F: '$1=="0"{print $3}' /proc/self/cgroup)"
Expand All @@ -174,7 +174,7 @@ jobs:
STAMP_EXIT=$?
if [ "$FLOOR_EXIT" -ne 0 ]; then exit "$FLOOR_EXIT"; fi
exit "$STAMP_EXIT"
timeout-minutes: 270
timeout-minutes: 55
- name: Floor cgroup peak post-read (calibration; survives a killed floor)
run: |
d="/sys/fs/cgroup$(awk -F: '$1=="0"{print $3}' /proc/self/cgroup)"
Expand Down
12 changes: 8 additions & 4 deletions dag/gunbc/ci_workflow.dag
Original file line number Diff line number Diff line change
Expand Up @@ -397,9 +397,13 @@ fn ci_deploy_step(stage: DeployStage) -> Step {

data gunbc_ci_job_timeout_policy_minutes: Int = 90

data gunbc_ci_floor_step_timeout_minutes: Duration = 270
data gunbc_ci_floor_step_timeout_minutes: Duration = 55

data gunbc_ci_floor_step_timeout_discovery_flip_note: String = "-> 270 restored 2026-07-12 on #6512 after 60m fail-fast regressed merge CI (receipt run 29197126623 @ 35f212fa5d: 60m step kill with 0 witness FAILs — batch-1 compile-clean PASS @ 15:01, batch-2 skip walk ~26m to recompute_trace_probe_test, then ~34m silent eval until kill; same serial-floor-wall class as 29183446733). Prior -> 60 operator ruling 2026-07-12: lower floor step cap from 270m so CI fails fast while the serial floor wall is debugged separately (receipt run 29183446733 @ 2e856a5617: 270m kill with 0 witness FAILs, batch-2 still in discovery SKIP walk). Prior -> 270 at PR #6464 receipt run 29151777611 (2026-07-11): 180m step cap still timed out mid batch 2 — 1535 SKIP rows finished @ 12:35:43, then ~147m silent witness-execution phase until the 180m kill @ 15:03:12 with batch 2 never completing (~177m in-batch from 12:06:38; ~443 non-skipped rows in roster of 1978). Prior -> 180 at run 29148344466 (~116m in-batch, ~87m post-skip). Prior -> 120 at run 29145270700. Prior -> 90 at run 29141663541. Prior -> 90 at body_lowering normalize hook (#6459 receipt run 29148735992 @ 3a372b7: floor step timed out at 60m MID discovery corpus after ~320 SKIP lines — batch-1 compile-clean normalize reconcile ~263s with body_lowering_fold in the normalize import closure; discovery still resolves every unique entry file before skip, so resolve->normalize pulled the ~1.3k-line scaffold into most closure walks). Structural fix on the same lane: NormalizedTree moved to v2.compiler.normalized_tree so v2.compiler.resolve no longer imports normalize (dissolve-on: skip-before-resolve so skipped rows never pay entry resolve). Prior step budget -> 60 at the discovery flip (gunbc.ci_spec ci_spec_discovery_flip_note; authored as 30 -> 60 on #6403, main had meanwhile bumped 30 -> 45 with the #6422 enrollments): the corpus discovery batch adds a whole-tree resolve plus the always-run live-tree rows to the floor step. Discovery corpus spawn_width_cap stays pinned to 1 (ci_corpus_discovery_spawn_width_cap in v2.workflow.ci_floor_plan): at width W the executor holds the parent's process-shared index PLUS W private shard indexes ((1+W) x whole-tree residency; width=2 OOM receipt run 28999086030), while width=1 runs rows on the main thread against the ONE shared index (union-resolve S1). CORRECTED RECEIPT (2026-07-10): run 29000557166's floor did NOT complete - the executor was host-OOM-killed ~8min in, MID discovery corpus (the log's later ExitSuccess belongs to the merge-admission STAMP tool, which stamped CI_FLOOR_EXIT=137); no flipped corpus has completed in CI yet - the only completion receipt is local (33.5GiB container, ~40min at width 5). The kill vector is host-level oversubscription (see gunbc_falsifier_plan_spawn_width_note), not this width model. #6475 receipt run 29143617420 @ 0c0f73: batch-1 compile-clean ~4m green; batch-2 discovery killed at 90m step cap mid-manual (last skip rust_wire_serde @ 07:47:49; ~15GiB peak). #6475 receipt run 29146814967 @ 6a61fce: same stall at 120m (last skip rust_wire_serde @ 09:34:49; ~85m silent eval). #6475 receipt run 29150477894 @ 6cd5326: same stall at 180m (last skip rust_wire_serde @ 11:43:46; ~88m silent eval) — local gunbc run on s1_closure_parses_holds reads 3/40 closure paths in 10m before timeout, matching the silent-eval class (live filesystem_read per path, not a skip-logged row). #6475 receipt run 29161709373 @ 9d9852c: batch-1 compile-clean refused — gunbc_ci_floor_step_timeout_discovery_flip_note string literal terminated early at col 2043 (body_lowering suffix spliced after closing quote during rebase merge); restored single-line literal on rebase to main #6464 note. Revisit down when floor memoization / resolver graph-major shrink the resolve wall; the affected-set selection receipts (skip counts per PR) are the cost dial to watch."
data gunbc_ci_regen_step_timeout_minutes: Duration = 15

data gunbc_ci_regen_step_timeout_note: String = "The regen step gets its OWN budget (operator main-timeout ruling 2026-07-23): it measured ~5min on every recent green run, and it previously borrowed the floor step's cap — which double-counted the floor budget in the job backstop sum and let a wedged regen sit for hours. 15m = 3x the measured envelope; a regen that exceeds it is a defect to diagnose, never headroom to grant."

data gunbc_ci_floor_step_timeout_discovery_flip_note: String = "-> 55 operator ruling 2026-07-23 (main wall ~75m: 'even 1 hour is absolutely ridiculous'): the 270 cap legalized a 4.5-hour crawl — receipt run 29976854620 @ 76fa6548e killed by hand at t=227m with current=16.3G pinned at memory.high, swap=34.4G, high_events=37,096,823, psi_some_avg10=33.82, oom_kill=0, governor 'hard back-off 1->1' every few seconds: the memory.high throttle-crawl class (retention finally exceeded the slot cap; the prior 40-50m greens were already pinned at 15.2-16.1G with zero headroom), and 8 subsequent main pushes wedged identically behind it fleet-wide. Recent green floors run 39-45m, so 55 is a real ceiling not a target; the regression ledger to hold onto: ~8m pre-#6848 -> ~18m (#6848) -> 40-50m (retention accreting to the cap) -> 4h (cap lost) — one disease (retention, not footprint), never an accepted baseline. Named follow-ups so this cap becomes the backstop rather than the diagnostic: the governor gains a terminal crawl-refusal arm (sustained high_events storm at width=1 -> typed FloorRefusedMemoryBudget naming batch/peak/swap, minutes not hours), and the #7106-attributed levers (one-tree-one-resolve, #6848 once-per-entry fixpoint, M2 eviction) bring the floor back under the cap with margin. Superseded main runs are NOT cancelled by policy (operator 2026-07-23): the per-commit verdict history is bisection evidence — the timeout IS the bound. PRIOR HISTORY: -> 270 restored 2026-07-12 on #6512 after 60m fail-fast regressed merge CI (receipt run 29197126623 @ 35f212fa5d: 60m step kill with 0 witness FAILs — batch-1 compile-clean PASS @ 15:01, batch-2 skip walk ~26m to recompute_trace_probe_test, then ~34m silent eval until kill; same serial-floor-wall class as 29183446733). Prior -> 60 operator ruling 2026-07-12: lower floor step cap from 270m so CI fails fast while the serial floor wall is debugged separately (receipt run 29183446733 @ 2e856a5617: 270m kill with 0 witness FAILs, batch-2 still in discovery SKIP walk). Prior -> 270 at PR #6464 receipt run 29151777611 (2026-07-11): 180m step cap still timed out mid batch 2 — 1535 SKIP rows finished @ 12:35:43, then ~147m silent witness-execution phase until the 180m kill @ 15:03:12 with batch 2 never completing (~177m in-batch from 12:06:38; ~443 non-skipped rows in roster of 1978). Prior -> 180 at run 29148344466 (~116m in-batch, ~87m post-skip). Prior -> 120 at run 29145270700. Prior -> 90 at run 29141663541. Prior -> 90 at body_lowering normalize hook (#6459 receipt run 29148735992 @ 3a372b7: floor step timed out at 60m MID discovery corpus after ~320 SKIP lines — batch-1 compile-clean normalize reconcile ~263s with body_lowering_fold in the normalize import closure; discovery still resolves every unique entry file before skip, so resolve->normalize pulled the ~1.3k-line scaffold into most closure walks). Structural fix on the same lane: NormalizedTree moved to v2.compiler.normalized_tree so v2.compiler.resolve no longer imports normalize (dissolve-on: skip-before-resolve so skipped rows never pay entry resolve). Prior step budget -> 60 at the discovery flip (gunbc.ci_spec ci_spec_discovery_flip_note; authored as 30 -> 60 on #6403, main had meanwhile bumped 30 -> 45 with the #6422 enrollments): the corpus discovery batch adds a whole-tree resolve plus the always-run live-tree rows to the floor step. Discovery corpus spawn_width_cap stays pinned to 1 (ci_corpus_discovery_spawn_width_cap in v2.workflow.ci_floor_plan): at width W the executor holds the parent's process-shared index PLUS W private shard indexes ((1+W) x whole-tree residency; width=2 OOM receipt run 28999086030), while width=1 runs rows on the main thread against the ONE shared index (union-resolve S1). CORRECTED RECEIPT (2026-07-10): run 29000557166's floor did NOT complete - the executor was host-OOM-killed ~8min in, MID discovery corpus (the log's later ExitSuccess belongs to the merge-admission STAMP tool, which stamped CI_FLOOR_EXIT=137); no flipped corpus has completed in CI yet - the only completion receipt is local (33.5GiB container, ~40min at width 5). The kill vector is host-level oversubscription (see gunbc_falsifier_plan_spawn_width_note), not this width model. #6475 receipt run 29143617420 @ 0c0f73: batch-1 compile-clean ~4m green; batch-2 discovery killed at 90m step cap mid-manual (last skip rust_wire_serde @ 07:47:49; ~15GiB peak). #6475 receipt run 29146814967 @ 6a61fce: same stall at 120m (last skip rust_wire_serde @ 09:34:49; ~85m silent eval). #6475 receipt run 29150477894 @ 6cd5326: same stall at 180m (last skip rust_wire_serde @ 11:43:46; ~88m silent eval) — local gunbc run on s1_closure_parses_holds reads 3/40 closure paths in 10m before timeout, matching the silent-eval class (live filesystem_read per path, not a skip-logged row). #6475 receipt run 29161709373 @ 9d9852c: batch-1 compile-clean refused — gunbc_ci_floor_step_timeout_discovery_flip_note string literal terminated early at col 2043 (body_lowering suffix spliced after closing quote during rebase merge); restored single-line literal on rebase to main #6464 note. Revisit down when floor memoization / resolver graph-major shrink the resolve wall; the affected-set selection receipts (skip counts per PR) are the cost dial to watch."

data gunbc_ci_step_timeout_measure_grounding_disposition: Disposition = Scaffold {
dissolves_to: SingleAuthority,
Expand Down Expand Up @@ -431,7 +435,7 @@ fn gunbc_ci_build_job_backstop_timeout_minutes() -> Int {
}

fn gunbc_ci_job_backstop_timeout_minutes() -> Int {
gunbc_ci_artifact_transfer_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_floor_step_timeout_minutes + gunbc_ci_floor_step_timeout_minutes + gunbc_ci_selection_control_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_prelude_allowance_minutes
gunbc_ci_artifact_transfer_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_floor_step_timeout_minutes + gunbc_ci_regen_step_timeout_minutes + gunbc_ci_selection_control_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_prelude_allowance_minutes
}

data gunbc_ci_job_timeout_policy_disposition: Disposition = Terminal {
Expand Down Expand Up @@ -506,7 +510,7 @@ fn ci_regen_floor_step() -> Step {
working_directory: none,
if_condition: none,
continue_on_error: none,
timeout_minutes: Present { value: gunbc_ci_floor_step_timeout_minutes }
timeout_minutes: Present { value: gunbc_ci_regen_step_timeout_minutes }
}
}

Expand Down
Loading