From 9e996e3febb5ccf98e0383f341d9432a647350af Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 23 Jul 2026 07:57:57 +0000 Subject: [PATCH] Main CI timeout: floor step 270 -> 55, regen gets its own 15m budget (job backstop derives 600 -> 130; effective main wall ~75m) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Operator ruling 2026-07-23 ('main needs a timeout, 75 seems reasonable for now; even 1 hour is absolutely ridiculous'), priced by today's receipt: run 29976854620 hand-killed at t=227m — 16.3G pinned at memory.high, 34.4G swap, 37,096,823 high_events, PSI 33.8, oom_kill=0, governor 'hard back-off 1->1' for hours — the memory.high throttle-crawl class, with 8 subsequent main pushes wedged identically behind it. The old 270m floor cap LEGALIZED a 4.5-hour crawl. Changes, all in the authority (ci.yml regenerated via main_wet, never hand-edited): - gunbc_ci_floor_step_timeout_minutes 270 -> 55 (recent green floors run 39-45m; 55 is a ceiling, not a target). The timeout-history note gains today's receipt plus the regression ledger (8m pre-#6848 -> 18m -> 40-50m cap-saturated -> 4h cap lost: one disease, retention — never an accepted baseline) and the named follow-ups (governor crawl-refusal arm; #7106 levers) so the cap becomes the backstop, not the diagnostic. - NEW gunbc_ci_regen_step_timeout_minutes = 15: the regen step (measured ~5m green) previously borrowed the floor's cap, double-counting the floor budget in the job backstop sum and allowing a wedged regen to sit for hours. - gunbc_ci_job_backstop_timeout_minutes() derives 600 -> 130 (sum with the regen term replacing the second floor term). Policy recorded on the carrier: superseded main runs are NOT cancelled (operator 2026-07-23) — per-commit verdict history is bisection evidence; the timeout is the bound. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_016fdkaGGLUKpLRwwqxp5sLg --- .github/workflows/ci.yml | 6 +++--- dag/gunbc/ci_workflow.dag | 12 ++++++++---- 2 files changed, 11 insertions(+), 7 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d56d10739df..e3b7f84b130 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -111,7 +111,7 @@ jobs: ci: runs-on: [self-hosted, linux, arm64] needs: [build] - timeout-minutes: 600 + timeout-minutes: 130 steps: - name: Checkout uses: actions/checkout@v5 @@ -149,7 +149,7 @@ jobs: fi fi "$ROOT/target/release/claim_executor" --source-root "$ROOT/dag" --source-root "$ROOT/src/v2" --plan-entry src/v2/workflow/ci_floor_plan.dag --plan-function gunbc_ci_regen_floor_batches --notice-title "self-host fixed-point (regen + staleness) — required; folded into ci job" - timeout-minutes: 270 + timeout-minutes: 15 - name: Floor cgroup peak pre-read (calibration; reset if permitted) run: | d="/sys/fs/cgroup$(awk -F: '$1=="0"{print $3}' /proc/self/cgroup)" @@ -174,7 +174,7 @@ jobs: STAMP_EXIT=$? if [ "$FLOOR_EXIT" -ne 0 ]; then exit "$FLOOR_EXIT"; fi exit "$STAMP_EXIT" - timeout-minutes: 270 + timeout-minutes: 55 - name: Floor cgroup peak post-read (calibration; survives a killed floor) run: | d="/sys/fs/cgroup$(awk -F: '$1=="0"{print $3}' /proc/self/cgroup)" diff --git a/dag/gunbc/ci_workflow.dag b/dag/gunbc/ci_workflow.dag index 2cbee7c8089..c28921d738a 100644 --- a/dag/gunbc/ci_workflow.dag +++ b/dag/gunbc/ci_workflow.dag @@ -397,9 +397,13 @@ fn ci_deploy_step(stage: DeployStage) -> Step { data gunbc_ci_job_timeout_policy_minutes: Int = 90 -data gunbc_ci_floor_step_timeout_minutes: Duration = 270 +data gunbc_ci_floor_step_timeout_minutes: Duration = 55 -data gunbc_ci_floor_step_timeout_discovery_flip_note: String = "-> 270 restored 2026-07-12 on #6512 after 60m fail-fast regressed merge CI (receipt run 29197126623 @ 35f212fa5d: 60m step kill with 0 witness FAILs — batch-1 compile-clean PASS @ 15:01, batch-2 skip walk ~26m to recompute_trace_probe_test, then ~34m silent eval until kill; same serial-floor-wall class as 29183446733). Prior -> 60 operator ruling 2026-07-12: lower floor step cap from 270m so CI fails fast while the serial floor wall is debugged separately (receipt run 29183446733 @ 2e856a5617: 270m kill with 0 witness FAILs, batch-2 still in discovery SKIP walk). Prior -> 270 at PR #6464 receipt run 29151777611 (2026-07-11): 180m step cap still timed out mid batch 2 — 1535 SKIP rows finished @ 12:35:43, then ~147m silent witness-execution phase until the 180m kill @ 15:03:12 with batch 2 never completing (~177m in-batch from 12:06:38; ~443 non-skipped rows in roster of 1978). Prior -> 180 at run 29148344466 (~116m in-batch, ~87m post-skip). Prior -> 120 at run 29145270700. Prior -> 90 at run 29141663541. Prior -> 90 at body_lowering normalize hook (#6459 receipt run 29148735992 @ 3a372b7: floor step timed out at 60m MID discovery corpus after ~320 SKIP lines — batch-1 compile-clean normalize reconcile ~263s with body_lowering_fold in the normalize import closure; discovery still resolves every unique entry file before skip, so resolve->normalize pulled the ~1.3k-line scaffold into most closure walks). Structural fix on the same lane: NormalizedTree moved to v2.compiler.normalized_tree so v2.compiler.resolve no longer imports normalize (dissolve-on: skip-before-resolve so skipped rows never pay entry resolve). Prior step budget -> 60 at the discovery flip (gunbc.ci_spec ci_spec_discovery_flip_note; authored as 30 -> 60 on #6403, main had meanwhile bumped 30 -> 45 with the #6422 enrollments): the corpus discovery batch adds a whole-tree resolve plus the always-run live-tree rows to the floor step. Discovery corpus spawn_width_cap stays pinned to 1 (ci_corpus_discovery_spawn_width_cap in v2.workflow.ci_floor_plan): at width W the executor holds the parent's process-shared index PLUS W private shard indexes ((1+W) x whole-tree residency; width=2 OOM receipt run 28999086030), while width=1 runs rows on the main thread against the ONE shared index (union-resolve S1). CORRECTED RECEIPT (2026-07-10): run 29000557166's floor did NOT complete - the executor was host-OOM-killed ~8min in, MID discovery corpus (the log's later ExitSuccess belongs to the merge-admission STAMP tool, which stamped CI_FLOOR_EXIT=137); no flipped corpus has completed in CI yet - the only completion receipt is local (33.5GiB container, ~40min at width 5). The kill vector is host-level oversubscription (see gunbc_falsifier_plan_spawn_width_note), not this width model. #6475 receipt run 29143617420 @ 0c0f73: batch-1 compile-clean ~4m green; batch-2 discovery killed at 90m step cap mid-manual (last skip rust_wire_serde @ 07:47:49; ~15GiB peak). #6475 receipt run 29146814967 @ 6a61fce: same stall at 120m (last skip rust_wire_serde @ 09:34:49; ~85m silent eval). #6475 receipt run 29150477894 @ 6cd5326: same stall at 180m (last skip rust_wire_serde @ 11:43:46; ~88m silent eval) — local gunbc run on s1_closure_parses_holds reads 3/40 closure paths in 10m before timeout, matching the silent-eval class (live filesystem_read per path, not a skip-logged row). #6475 receipt run 29161709373 @ 9d9852c: batch-1 compile-clean refused — gunbc_ci_floor_step_timeout_discovery_flip_note string literal terminated early at col 2043 (body_lowering suffix spliced after closing quote during rebase merge); restored single-line literal on rebase to main #6464 note. Revisit down when floor memoization / resolver graph-major shrink the resolve wall; the affected-set selection receipts (skip counts per PR) are the cost dial to watch." +data gunbc_ci_regen_step_timeout_minutes: Duration = 15 + +data gunbc_ci_regen_step_timeout_note: String = "The regen step gets its OWN budget (operator main-timeout ruling 2026-07-23): it measured ~5min on every recent green run, and it previously borrowed the floor step's cap — which double-counted the floor budget in the job backstop sum and let a wedged regen sit for hours. 15m = 3x the measured envelope; a regen that exceeds it is a defect to diagnose, never headroom to grant." + +data gunbc_ci_floor_step_timeout_discovery_flip_note: String = "-> 55 operator ruling 2026-07-23 (main wall ~75m: 'even 1 hour is absolutely ridiculous'): the 270 cap legalized a 4.5-hour crawl — receipt run 29976854620 @ 76fa6548e killed by hand at t=227m with current=16.3G pinned at memory.high, swap=34.4G, high_events=37,096,823, psi_some_avg10=33.82, oom_kill=0, governor 'hard back-off 1->1' every few seconds: the memory.high throttle-crawl class (retention finally exceeded the slot cap; the prior 40-50m greens were already pinned at 15.2-16.1G with zero headroom), and 8 subsequent main pushes wedged identically behind it fleet-wide. Recent green floors run 39-45m, so 55 is a real ceiling not a target; the regression ledger to hold onto: ~8m pre-#6848 -> ~18m (#6848) -> 40-50m (retention accreting to the cap) -> 4h (cap lost) — one disease (retention, not footprint), never an accepted baseline. Named follow-ups so this cap becomes the backstop rather than the diagnostic: the governor gains a terminal crawl-refusal arm (sustained high_events storm at width=1 -> typed FloorRefusedMemoryBudget naming batch/peak/swap, minutes not hours), and the #7106-attributed levers (one-tree-one-resolve, #6848 once-per-entry fixpoint, M2 eviction) bring the floor back under the cap with margin. Superseded main runs are NOT cancelled by policy (operator 2026-07-23): the per-commit verdict history is bisection evidence — the timeout IS the bound. PRIOR HISTORY: -> 270 restored 2026-07-12 on #6512 after 60m fail-fast regressed merge CI (receipt run 29197126623 @ 35f212fa5d: 60m step kill with 0 witness FAILs — batch-1 compile-clean PASS @ 15:01, batch-2 skip walk ~26m to recompute_trace_probe_test, then ~34m silent eval until kill; same serial-floor-wall class as 29183446733). Prior -> 60 operator ruling 2026-07-12: lower floor step cap from 270m so CI fails fast while the serial floor wall is debugged separately (receipt run 29183446733 @ 2e856a5617: 270m kill with 0 witness FAILs, batch-2 still in discovery SKIP walk). Prior -> 270 at PR #6464 receipt run 29151777611 (2026-07-11): 180m step cap still timed out mid batch 2 — 1535 SKIP rows finished @ 12:35:43, then ~147m silent witness-execution phase until the 180m kill @ 15:03:12 with batch 2 never completing (~177m in-batch from 12:06:38; ~443 non-skipped rows in roster of 1978). Prior -> 180 at run 29148344466 (~116m in-batch, ~87m post-skip). Prior -> 120 at run 29145270700. Prior -> 90 at run 29141663541. Prior -> 90 at body_lowering normalize hook (#6459 receipt run 29148735992 @ 3a372b7: floor step timed out at 60m MID discovery corpus after ~320 SKIP lines — batch-1 compile-clean normalize reconcile ~263s with body_lowering_fold in the normalize import closure; discovery still resolves every unique entry file before skip, so resolve->normalize pulled the ~1.3k-line scaffold into most closure walks). Structural fix on the same lane: NormalizedTree moved to v2.compiler.normalized_tree so v2.compiler.resolve no longer imports normalize (dissolve-on: skip-before-resolve so skipped rows never pay entry resolve). Prior step budget -> 60 at the discovery flip (gunbc.ci_spec ci_spec_discovery_flip_note; authored as 30 -> 60 on #6403, main had meanwhile bumped 30 -> 45 with the #6422 enrollments): the corpus discovery batch adds a whole-tree resolve plus the always-run live-tree rows to the floor step. Discovery corpus spawn_width_cap stays pinned to 1 (ci_corpus_discovery_spawn_width_cap in v2.workflow.ci_floor_plan): at width W the executor holds the parent's process-shared index PLUS W private shard indexes ((1+W) x whole-tree residency; width=2 OOM receipt run 28999086030), while width=1 runs rows on the main thread against the ONE shared index (union-resolve S1). CORRECTED RECEIPT (2026-07-10): run 29000557166's floor did NOT complete - the executor was host-OOM-killed ~8min in, MID discovery corpus (the log's later ExitSuccess belongs to the merge-admission STAMP tool, which stamped CI_FLOOR_EXIT=137); no flipped corpus has completed in CI yet - the only completion receipt is local (33.5GiB container, ~40min at width 5). The kill vector is host-level oversubscription (see gunbc_falsifier_plan_spawn_width_note), not this width model. #6475 receipt run 29143617420 @ 0c0f73: batch-1 compile-clean ~4m green; batch-2 discovery killed at 90m step cap mid-manual (last skip rust_wire_serde @ 07:47:49; ~15GiB peak). #6475 receipt run 29146814967 @ 6a61fce: same stall at 120m (last skip rust_wire_serde @ 09:34:49; ~85m silent eval). #6475 receipt run 29150477894 @ 6cd5326: same stall at 180m (last skip rust_wire_serde @ 11:43:46; ~88m silent eval) — local gunbc run on s1_closure_parses_holds reads 3/40 closure paths in 10m before timeout, matching the silent-eval class (live filesystem_read per path, not a skip-logged row). #6475 receipt run 29161709373 @ 9d9852c: batch-1 compile-clean refused — gunbc_ci_floor_step_timeout_discovery_flip_note string literal terminated early at col 2043 (body_lowering suffix spliced after closing quote during rebase merge); restored single-line literal on rebase to main #6464 note. Revisit down when floor memoization / resolver graph-major shrink the resolve wall; the affected-set selection receipts (skip counts per PR) are the cost dial to watch." data gunbc_ci_step_timeout_measure_grounding_disposition: Disposition = Scaffold { dissolves_to: SingleAuthority, @@ -431,7 +435,7 @@ fn gunbc_ci_build_job_backstop_timeout_minutes() -> Int { } fn gunbc_ci_job_backstop_timeout_minutes() -> Int { - gunbc_ci_artifact_transfer_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_floor_step_timeout_minutes + gunbc_ci_floor_step_timeout_minutes + gunbc_ci_selection_control_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_prelude_allowance_minutes + gunbc_ci_artifact_transfer_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_floor_step_timeout_minutes + gunbc_ci_regen_step_timeout_minutes + gunbc_ci_selection_control_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_aux_step_timeout_minutes + gunbc_ci_prelude_allowance_minutes } data gunbc_ci_job_timeout_policy_disposition: Disposition = Terminal { @@ -506,7 +510,7 @@ fn ci_regen_floor_step() -> Step { working_directory: none, if_condition: none, continue_on_error: none, - timeout_minutes: Present { value: gunbc_ci_floor_step_timeout_minutes } + timeout_minutes: Present { value: gunbc_ci_regen_step_timeout_minutes } } }