diff --git a/bin/fm-crew-state.sh b/bin/fm-crew-state.sh index 1512cf83c0a..a85d697d0d6 100755 --- a/bin/fm-crew-state.sh +++ b/bin/fm-crew-state.sh @@ -37,13 +37,15 @@ # branch (branch_sync.state=pipeline_owned), its own custody attribution # binds an ACTIVE run without head equality (fm_nm_run_is_pipeline_owned_active # in bin/fm-nm-run-lib.sh). -# A run head whose commit object the task copy never fetched (the pipeline -# committed its fix round in its own checkout) cannot be verified locally; +# A run head the task copy cannot bind - never fetched (the pipeline +# committed its fix round in its own checkout), or resolvable but off this +# worktree's line of history (the pipeline replayed the branch onto an +# advanced upstream) - cannot be verified locally; # that row is recognized only as a provable pipeline-owned continuation - # the branch's ACTIVE newest ledger row, anchored by the row immediately # before it having ended at exactly this worktree's head - so an active fix # round never reads as an older failed run (rule owned by -# fm_nm_runs_status_for_worktree in bin/fm-nm-run-lib.sh). +# fm_nm_runs_decision_for_worktree in bin/fm-nm-run-lib.sh). # More than one recorded run can bind to this worktree at once, and # bin/fm-nm-run-lib.sh also owns which of them wins: a LIVE run always # outranks a terminal one, so a terminal answer here is provisional until @@ -118,8 +120,13 @@ META=${FM_CREW_STATE_META_OVERRIDE:-"$STATE/$ID.meta"} LOG=${FM_CREW_STATE_STATUS_OVERRIDE:-"$STATE/$ID.status"} NM_TIMEOUT=${FM_CREW_STATE_NM_TIMEOUT:-10} case "$NM_TIMEOUT" in ''|*[!0-9]*) NM_TIMEOUT=10 ;; esac +# Bound for the best-effort calls of nm_inspect_live_run below - the home view +# and the per-run inspection - deliberately far shorter than the authoritative +# reads they enrich, because neither can do more than add detail to a word the +# ledger has already decided. +NM_INSPECT_TIMEOUT=3 # How many of the most recent `no-mistakes runs` rows each ledger read -# (fm_nm_runs_status_for_worktree in bin/fm-nm-run-lib.sh) scans, whether it is +# (fm_nm_runs_decision_for_worktree in bin/fm-nm-run-lib.sh) scans, whether it is # the cross-branch fallback or the live-sibling probe behind a terminal `axi # status` answer (docs/configuration.md owns the setting). Generous enough to # still find a branch's own run on a busy multi-crew fleet without listing the @@ -533,7 +540,7 @@ nm_ci_checks_state() { # run-listing command is the top-level `no-mistakes runs` (the `axi` surface # has no runs-listing subcommand; tests/fm-crew-state.test.sh owns the # 2026-07-02 dead-code incident history this fallback replaced). -# fm_nm_runs_status_for_worktree in bin/fm-nm-run-lib.sh is the ONE owner of +# fm_nm_runs_decision_for_worktree in bin/fm-nm-run-lib.sh is the ONE owner of # the ledger format, the newest-row-decides rule, its live-over-terminal # exception, and the anchored pipeline-continuation recognition # (model-routing-benchmark-hardening: an active fix round whose head object the @@ -558,6 +565,55 @@ nm_run_head_matches_worktree() { fm_nm_head_matches_worktree "$WT" "$run_head" } +# Recover the live run's own step and gate detail once the ledger has attributed +# that run to this worktree. The ledger has no run id column +# (bin/fm-nm-run-lib.sh), so the coarse path below can only report the status +# WORD `running`, which reads as work in progress even when the run is sitting +# at a gate nobody has answered: the 2026-09-15 incident reported a run parked +# 23m57s at fix_review with an ask-user finding as `validating`, and +# bin/fm-fleet-snapshot.sh counts `working` as active work while only +# `parked`/`paused`/`blocked` reach its waiting list, so the crew could have sat +# there indefinitely with nobody told a decision was owed. The id comes from the +# recent-runs table of the answer already in hand when it carries one, else from +# one home-view call, and only then is that ONE run inspected by id. +# This recovers detail for a run the ledger already attributed; it never widens +# attribution. Every step must prove itself - the id must name the attributed +# row, the answer must be that id, on this crew's branch, and still live - and +# anything unproven leaves the coarse word exactly as the ledger decided it. +nm_inspect_live_run() { # + local row_head=$1 id detail + [ -n "$row_head" ] || return 1 + # Both id-bearing surfaces expose only the ten most recently created runs for + # the repo, while the ledger that attributed this run scans far more rows, so + # a run parked longer than ten newer runs take to appear has no recoverable + # id and keeps the coarse word. No id-bearing listing accepts a caller-chosen + # limit, so closing this needs an upstream surface that does, or reading the + # tool's internal state directly. + id=$(fm_nm_home_view_run_id "$RUN_OUT" "$CREW_BRANCH" "$row_head") + # Both calls below are pure enrichment, and both are subprocesses of the + # installed CLI, which runs its own network update check on every invocation + # (unbounded by anything local when the network is slow). They share the short + # bound above, because missing the gate detail costs one coarse word the + # ledger already proved, while a slow enrichment would slow every supervision + # read that reaches for it. + [ -n "$id" ] \ + || id=$(fm_nm_home_view_run_id \ + "$(fm_nm_run "$WT" "$NM_INSPECT_TIMEOUT" axi)" "$CREW_BRANCH" "$row_head") + [ -n "$id" ] || return 1 + detail=$(fm_nm_run "$WT" "$NM_INSPECT_TIMEOUT" axi status --run "$id") + [ -n "$detail" ] || return 1 + # `axi status --run` renders another branch's run under `other_branch_run:`, + # which never answers for this worktree, and a terminal answer must never + # displace the live word the ledger proved: manufacturing a terminal verdict + # out of a second query is the same false-failure class the live-over-terminal + # rule exists to prevent. + [ "$(fm_nm_strip_quotes "$(fm_nm_field "$detail" id)")" = "$id" ] || return 1 + [ "$(fm_nm_strip_quotes "$(fm_nm_field "$detail" branch)")" = "$CREW_BRANCH" ] || return 1 + fm_nm_run_is_active "$detail" || return 1 + RUN_OUT=$detail + RUN_SOURCE=full +} + HAVE_RUN=0 # RUN_SOURCE distinguishes the two ways HAVE_RUN=1 can happen: "full" means # $RUN_OUT is real `axi status` TOON with step/gate detail (including a @@ -589,10 +645,15 @@ if [ "$KIND" = ship ] && [ -n "$CREW_BRANCH" ] && command -v no-mistakes >/dev/n # displaces it: a terminal run with no live sibling keeps its full # `axi status` step and gate detail rather than degrading to the ledger. if ! fm_nm_run_is_active "$RUN_OUT"; then - live_status=$(fm_nm_runs_status_for_worktree "$WT" "$CREW_BRANCH" "$(nm_runs_list)") + live_decision=$(fm_nm_runs_decision_for_worktree "$WT" "$CREW_BRANCH" "$(nm_runs_list)") + live_status=${live_decision%% *} if [ "$(fm_nm_run_status_class "$live_status")" = live ]; then COARSE_STATUS=$live_status RUN_SOURCE=coarse + # The ledger proved the live run; its gate is only readable by + # inspecting that run itself, and failing to reach it keeps the + # coarse word. + nm_inspect_live_run "${live_decision#* }" || true fi fi else @@ -602,15 +663,22 @@ if [ "$KIND" = ship ] && [ -n "$CREW_BRANCH" ] && command -v no-mistakes >/dev/n # `[ -n "$RUN_OUT" ]`: an empty/timed-out primary call means the CLI # itself did not respond, so retrying it immediately with a second # bounded call would just double the wait for no better answer. - COARSE_STATUS=$(fm_nm_runs_status_for_worktree "$WT" "$CREW_BRANCH" "$(nm_runs_list)") + coarse_decision=$(fm_nm_runs_decision_for_worktree "$WT" "$CREW_BRANCH" "$(nm_runs_list)") + COARSE_STATUS=${coarse_decision%% *} if [ -n "$COARSE_STATUS" ]; then HAVE_RUN=1 # A branch-matching answer the strict rule rejected is this branch's # own current run once the ledger proves the pipeline-owned # continuation, so its axi TOON is the authoritative run detail # (RUN_SOURCE stays full); only a foreign-branch answer leaves - # coarse status-word detail. - [ "$run_branch" = "$CREW_BRANCH" ] || RUN_SOURCE=coarse + # coarse status-word detail, and a live one is inspected for its own + # gate rather than reported as the bare word `running`. + if [ "$run_branch" != "$CREW_BRANCH" ]; then + RUN_SOURCE=coarse + if [ "$(fm_nm_run_status_class "$COARSE_STATUS")" = live ]; then + nm_inspect_live_run "${coarse_decision#* }" || true + fi + fi fi fi fi diff --git a/bin/fm-nm-run-lib.sh b/bin/fm-nm-run-lib.sh index ed71d315fd8..2b90715d705 100644 --- a/bin/fm-nm-run-lib.sh +++ b/bin/fm-nm-run-lib.sh @@ -5,7 +5,7 @@ # fm-crew-state.sh (read-only current-state reporting) and fm-teardown.sh # (pre-teardown run abort, see its "Fix 1" header comment). Both bind a run # by strict branch-and-head identity first, and both then recognize a provable -# pipeline-owned continuation through fm_nm_runs_status_for_worktree below: +# pipeline-owned continuation through fm_nm_runs_decision_for_worktree below: # crew-state for an ACTIVE run, so a fix round never reads as an older failed # run, and teardown for a run PARKED at a gate, so cleanup concludes it # instead of orphaning it. Getting this wrong in either @@ -75,7 +75,7 @@ fm_nm_resolve_commit() { # # - run head is a strict ancestor of worktree HEAD, or diverged: no match # (local work advanced outside the run, or the branch tip was rewritten) # A run head whose object this copy does not have cannot be proven here and is -# rejected; fm_nm_runs_status_for_worktree below owns the one ledger-anchored +# rejected; fm_nm_runs_decision_for_worktree below owns the one ledger-anchored # recognition for that case, and fm_nm_run_is_pipeline_owned_active below # carries the custody exemption: a live run whose pipeline currently owns the # branch binds without head equality. @@ -90,7 +90,7 @@ fm_nm_resolve_commit() { # # match rule each one used, because a terminal run can be the corpse of a # crashed attempt while the live one is what is actually validating this code. # Within one liveness class the selecting caller's existing precedence is -# unchanged - for the runs ledger, fm_nm_runs_status_for_worktree's +# unchanged - for the runs ledger, fm_nm_runs_decision_for_worktree's # newest-row-decides rule below. # fm_nm_run_status_class next classifies a recorded status word for that # comparison, and a word it cannot classify keeps the caller's own precedence @@ -159,26 +159,34 @@ fm_nm_run_is_pipeline_owned_active() { # # per-row scan-and-skip. The ledger is the real top-level `no-mistakes runs # --limit N` listing (plain text, no run id, no quoting, newest-first, columns # " []"; the `axi` surface has no -# runs-listing subcommand - verified against the installed CLI). Prints the -# status word of the branch's CURRENT run row, or nothing when the ledger -# cannot prove attribution. When optional expected head $4 is supplied, its +# runs-listing subcommand, only the home view `no-mistakes axi` prints with no +# subcommand at all, which fm_nm_home_view_run_id below reads for the run ids +# this ledger omits - both verified against the installed CLI, v1.72.0). Prints +# " " for the branch's CURRENT run row, or nothing when the +# ledger cannot prove attribution. When optional expected head $4 is supplied, its # abbreviated commit identity must match the newest row. The branch's NEWEST # row alone decides; older rows are history and never answer for the present: # - newest row's head resolves and matches the worktree (fm_nm_head_matches_worktree): # its status word -# - newest row's head resolves but does not match: nothing (a newer run that -# is not this worktree's makes every older row stale history) -# - newest row's head does not resolve in this copy (the pipeline committed -# its fix round in its own checkout and the task copy never fetched it): -# recognized ONLY as a provable pipeline-owned continuation of the -# submitted head, which requires ALL of: the row is ACTIVE (status -# running), and the immediately older row for the SAME branch resolves to -# EXACTLY the worktree HEAD. The pipeline's own ledger then proves an -# unbroken run sequence from a run that ended at the submitted head to an -# active run on the same branch - the anchored active row's status word is -# printed. Anything else (no anchor row, an anchor that is merely an -# ancestor, a terminal unresolvable row) prints nothing, so branch-name -# coincidence, arbitrary remote state, and other tasks' runs never match. +# - newest row's head does not bind - either it does not resolve in this copy +# (the pipeline committed its fix round in its own checkout and the task +# copy never fetched it) or it resolves on a line of history the worktree +# HEAD does not share (the pipeline replayed the branch onto an advanced +# upstream, so neither commit descends from the other): +# a TERMINAL or unclassifiable row prints nothing, because a newer run that +# is not this worktree's makes every older row stale history. An ACTIVE row +# is recognized as a provable pipeline-owned continuation of the submitted +# head, which additionally requires the immediately older row for the SAME +# branch to resolve to EXACTLY the worktree HEAD. The pipeline's own ledger +# then proves an unbroken run sequence from a run that ended at the +# submitted head to an active run on the same branch - the anchored active +# row's status word is printed. Anything else (no anchor row, an anchor +# that is merely an ancestor or merely a descendant) prints nothing, so +# branch-name coincidence, arbitrary remote state, and other tasks' runs +# never match. Both unbindable shapes reach the SAME anchor because a +# rebased head is exactly as unprovable as an unfetched one, and treating +# only the unfetched one that way is what let a dead run report a live +# task as failed (nutrifam-cerrar-allow-authenticated, 2026-09-07). # The one exception to newest-row-decides is the live-over-terminal rule stated # with fm_nm_head_matches_worktree above, and it only ever replaces a TERMINAL # answer with a LIVE one: when the newest row binds but is terminal, the older @@ -191,20 +199,21 @@ fm_nm_run_is_pipeline_owned_active() { # # requires, so branch-name coincidence and other tasks' runs still never # match. A terminal newest row is the corpse of a crashed attempt whenever a # live run for the same worktree is still on the ledger, so it is not the -# present. Nothing else widens: a newest row that does not bind still ends the -# scan, a newest row whose class is live or unclassifiable is still answered -# as-is, the anchored pipeline-continuation path is untouched, and with no live -# sibling the newest terminal word is still what is printed. +# present. Nothing else widens: the sibling scan is unchanged, a newest row +# that does not bind still ends the scan unless it is LIVE and its head is +# unprovable rather than superseded, a newest row that binds is still answered +# as-is, the anchor is still exact head equality and nothing else, and with no +# live sibling the newest terminal word is still what is printed. # Read-only: git reads resolve objects in place; custody never changes. -fm_nm_runs_status_for_worktree() { # [expected-head] +fm_nm_runs_decision_for_worktree() { # [expected-head] local wt=$1 branch=$2 list=$3 expected_head=${4:-} - local local_full row_full row st br sha day clock pr extra year_num month_num day_num max_day pending_st='' + local local_full row_full row st br sha day clock pr extra year_num month_num day_num max_day pending_st='' pending_sha='' # Set only by the newest binding row when its status classifies terminal, and # printed when the scan ends without finding a live row for this worktree. It # is the sole reason the scan continues past the newest row, and every exit # below leaves the loop rather than returning, so a malformed older row can # never swallow an answer the newest row had already decided. - local decided='' decided_exact='' + local decided='' decided_exact='' decided_sha='' local_full=$(git -C "$wt" rev-parse HEAD 2>/dev/null) || return 0 [ -n "$list" ] || return 0 while IFS= read -r row; do @@ -249,6 +258,7 @@ fm_nm_runs_status_for_worktree() { # [ex [ -n "$decided_exact" ] || continue fi decided=$st + decided_sha=$sha break fi if [ -n "$pending_st" ]; then @@ -257,6 +267,7 @@ fm_nm_runs_status_for_worktree() { # [ex # worktree still sits at the submitted head. if [ "$(fm_nm_resolve_commit "$wt" "$sha")" = "$local_full" ]; then decided=$pending_st + decided_sha=$pending_sha fi break fi @@ -269,21 +280,94 @@ fm_nm_runs_status_for_worktree() { # [ex esac fi row_full=$(fm_nm_resolve_commit "$wt" "$sha") - if [ -n "$row_full" ]; then - if fm_nm_head_matches_worktree "$wt" "$sha"; then - decided=$st - # A live or unclassifiable word is this worktree's current answer and - # ends the scan; only a terminal one keeps looking for a live sibling. - if [ "$(fm_nm_run_status_class "$st")" = terminal ]; then - [ "$row_full" != "$local_full" ] || decided_exact=1 - continue - fi + if [ -n "$row_full" ] && fm_nm_head_matches_worktree "$wt" "$sha"; then + decided=$st + decided_sha=$sha + # A live or unclassifiable word is this worktree's current answer and + # ends the scan; only a terminal one keeps looking for a live sibling. + if [ "$(fm_nm_run_status_class "$st")" = terminal ]; then + [ "$row_full" != "$local_full" ] || decided_exact=1 + continue fi break fi - [ "$st" = running ] || break + # The head rule could not bind this row. A head that resolves as a strict + # ANCESTOR of the worktree HEAD is not unprovable, it is superseded: local + # work advanced past it outside the run (the case fm_nm_head_matches_worktree + # rejects on purpose), and no anchor can turn stale history into the present. + if [ -n "$row_full" ] \ + && git -C "$wt" merge-base --is-ancestor "$row_full" "$local_full" 2>/dev/null; then + break + fi + # What remains is genuinely unprovable: a head absent from this copy, or one + # on a line of history the worktree HEAD does not share. Only a LIVE row is + # still recognizable, through the anchor below. + [ "$(fm_nm_run_status_class "$st")" = live ] || break pending_st=$st + pending_sha=$sha done <<< "$list" - printf '%s' "$decided" + if [ -n "$decided" ]; then + printf '%s %s' "$decided" "$decided_sha" + fi return 0 } + +# The run id of the live run the ledger already attributed to this worktree, +# read from a recent-runs TOON table in captured output $1 (the home view +# `no-mistakes axi` prints with no subcommand, and the same table a status +# response carries). The ledger the decision above scans has no id column, so +# this is the only public way to name that run for `axi status --run `. +# Deliberately NOT an attribution rule of its own: it never selects a run, it +# only recovers the id of the row the caller already proved is this worktree's, +# so branch $2 AND recorded head $3 must both match and the row's status word +# must still classify live. The head comparison is the ledger's own abbreviated +# form (either identity a prefix of the other), because the two surfaces +# abbreviate independently, and an empty or non-hex head column is no head at +# all rather than a match against everything. Column positions come from the +# table's own header rather than a fixed order, branch and status are compared +# against exact expected words, and the head and the id are charset-validated, +# so a reshaped or malformed table yields nothing instead of a wrong id. +fm_nm_home_view_run_id() { # + local out=$1 branch=$2 head=$3 id + [ -n "$out" ] && [ -n "$branch" ] && [ -n "$head" ] || return 0 + id=$(printf '%s\n' "$out" | awk -v want_branch="$branch" -v want_head="$head" ' + function unquote(v) { gsub(/^[ \t]+|[ \t]+$/, "", v); gsub(/^"|"$/, "", v); return v } + /^[[:space:]]*runs\[[0-9]+\]\{[^}]*\}:[[:space:]]*$/ { + header = $0 + sub(/^[^{]*\{/, "", header) + sub(/\}.*$/, "", header) + n = split(header, cols, ",") + id_col = branch_col = status_col = head_col = 0 + for (i = 1; i <= n; i++) { + col = unquote(cols[i]) + if (col == "id") id_col = i + else if (col == "branch") branch_col = i + else if (col == "status") status_col = i + else if (col == "head") head_col = i + } + in_table = (id_col && branch_col && status_col && head_col) + next + } + in_table { + if ($0 !~ /^[[:space:]]+[^[:space:]]/ || split($0, f, ",") < n) { in_table = 0; next } + if (unquote(f[branch_col]) != want_branch) next + if (unquote(f[status_col]) != "running") next + row_head = unquote(f[head_col]) + if (row_head == "" || row_head ~ /[^A-Fa-f0-9]/) next + if (index(row_head, want_head) != 1 && index(want_head, row_head) != 1) next + print unquote(f[id_col]) + exit + } + ') + case "$id" in *[!A-Za-z0-9]*|'') return 0 ;; esac + printf '%s' "$id" +} + +# The status word alone, the historical contract every caller but crew-state's +# gate inspection uses: the first field of the decision above, so one scan +# serves both. +fm_nm_runs_status_for_worktree() { # [expected-head] + local decision + decision=$(fm_nm_runs_decision_for_worktree "$@") + printf '%s' "${decision%% *}" +} diff --git a/docs/architecture.md b/docs/architecture.md index 067c92618e2..f23c3fdabec 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -79,11 +79,16 @@ Any direct or remaining historical annotation prints every status line unread at For other daemon, timeout, or unreachability claims, a running or fixing run with recent pipeline-reported activity supersedes the event and names reattachment as the recovery instead of surfacing a false block. [`bin/fm-nm-run-lib.sh`](../bin/fm-nm-run-lib.sh)'s header owns the exact branch, head, pipeline-custody, and newest-first attribution rules. It also owns which binding run wins when more than one recorded run binds to the same worktree: a live run outranks a terminal one, so a crashed run sitting at the worktree's own commit never reports a healthy task as failed while its live successor is still validating. -A run head the task copy cannot resolve locally is attributed only when the pipeline's own runs ledger proves it is an active continuation of the submitted head, so a pipeline fix round never reads as an older failed run. +A newest run head the task copy cannot bind, whether it never fetched the commit or the pipeline replayed the branch onto an advanced upstream, is attributed only when the pipeline's own runs ledger proves it is an active continuation of the submitted head, so neither a pipeline fix round nor a live run the pipeline rebased ahead of a dead one ever reads as that older failed run. +The live-over-terminal search across older rows stays narrower than that, on the terms the header states. +A run head this copy has already advanced past is superseded local history rather than an unprovable one, so no ledger proof attributes it. During no-mistakes' `ci` monitor phase, it also reads the ci step log tail because `axi status` reports both "still waiting on checks" and "checks green, waiting on merge" as `ci,running`. The most recent recognized ci log marker wins, so checks-green monitoring reports done while a later re-arm, failed-check, or issue marker returns the crew to working. A terminal failed run whose only failure is the ci monitor step, after every substantive step completed and the same marker reads checks green, also reports done with the run's PR URL, because a monitor whose only remaining job is to observe a human merge decision must not convert the absence of that decision into a failure verdict. In the coarse runs-ledger fallback, which has no steps table and no ci log, a terminal failed record whose daemon an explicit `daemon status` probe proves down reports unknown as unverified instead: an instrument failure must never read as work failure. +Before a coarse LIVE word stands as the answer, that one run is inspected by run id - taken from the recent-runs table of the answer already in hand, else from one bounded home-view call, because the ledger itself records no run id - so a run parked at a gate nobody has answered reports that gate instead of reading as work in progress, which the fleet view counts as active rather than as a wait. +That recovery holds while the attributed run is still among the ten most recently created runs the id-bearing surfaces expose, which is far fewer rows than the ledger itself scans; past that horizon the reading falls back to the coarse word it had before. +The inspection recovers detail for the run the ledger already attributed and never widens attribution: the id must name that attributed row, the answer must be that id on this crew's branch and still live, and anything it cannot prove leaves the ledger's own word standing. Only when no matching run exists does it consult semantic busy state; exact busy reports working, exact idle permits fallback to a status-log event whose verb maps to a recognized run-state, and unknown or a dead pane stays unknown instead of trusting a stale log. Decision-only events such as `resolved` never become current state or leak their prose into the current-state detail. In that status-log fallback, a declared external wait reports the distinct `paused` state with its reason. diff --git a/docs/configuration.md b/docs/configuration.md index e9735f0fde9..15cbce93a7b 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -1022,7 +1022,7 @@ FM_PROCEVENT_LAUNCH_FLOOR_SECONDS=1 # minimum interval between launches of o FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS=3 # how long reconcile waits for the runners it started to prove they are running; 1..600, keep well below FM_POLL FM_WHEN_OUTPUT_TAIL_BYTES=8192 # bound on the command-output tail inside one condition->action outcome document FM_CODEX_WATCH_CHECKPOINT=180 # seconds per foreground watcher checkpoint in Codex primary supervision -FM_CREW_STATE_NM_TIMEOUT=10 # seconds allowed per no-mistakes query inside fm-crew-state.sh +FM_CREW_STATE_NM_TIMEOUT=10 # seconds allowed per authoritative no-mistakes query inside fm-crew-state.sh; the live run's best-effort gate inspection (home view, per-run status) keeps its own fixed 3-second bound so a slow enrichment cannot stretch a supervision read FM_TEARDOWN_NM_TIMEOUT=10 # seconds allowed per no-mistakes query or abort inside fm-teardown.sh FM_CREW_STATE_RUNS_LIMIT=200 # recent no-mistakes run rows scanned when the runs ledger is consulted: axi status cannot be attributed directly, or its answer is terminal and may have a live sibling run FM_TEARDOWN_NM_RUNS_LIMIT=200 # recent no-mistakes run rows scanned to prove an unresolved-head parked run belongs to teardown's task diff --git a/tests/fm-crew-state.test.sh b/tests/fm-crew-state.test.sh index da93917667d..f978633cd78 100755 --- a/tests/fm-crew-state.test.sh +++ b/tests/fm-crew-state.test.sh @@ -23,8 +23,10 @@ # (e) cross-branch attribution: this branch's own run found via list lookup # (e2) several runs bound to one worktree: the live one outranks the corpse # (an unclassifiable status word keeps the ledger's newest-first order) -# (e3) the live sibling's head was never fetched into the task copy: it still -# outranks a terminal row sitting at the worktree's exact commit +# (e3) the live sibling's head cannot be bound - never fetched into the task +# copy, or replayed onto an advanced upstream by the pipeline: it still +# outranks a terminal row sitting at the worktree's exact commit, and +# without that exact anchor it binds nothing at all # (f) no run + semantic busy -> pane # (g) no run + semantic idle falls to the status-log verb -> status-log # (h) dead pane: no run -> unknown/none; with a run -> run-step (not the shell) @@ -83,10 +85,24 @@ case "${1:-}" in case "${1:-}" in status) shift - if [ "${1:-}" = --run ]; then printf '%s\n' "${FM_FAKE_AXI_STATUS_RUN:-}" + if [ "${1:-}" = --run ]; then + # FM_FAKE_AXI_RUN_ID pins WHICH run id the fixture answers for, the + # way the real CLI answers only the id that exists: any other id gets + # the error response, so a test can prove the helper asked for the + # live run and not some other row. + if [ -n "${FM_FAKE_AXI_RUN_ID:-}" ] && [ "${2:-}" != "${FM_FAKE_AXI_RUN_ID}" ]; then + printf 'error: run %s not found\n' "${2:-}" + else + printf '%s\n' "${FM_FAKE_AXI_STATUS_RUN:-}" + fi else printf '%s\n' "${FM_FAKE_AXI_STATUS:-}"; fi ;; logs) printf '%s\n' "${FM_FAKE_CI_LOGS:-}" ;; + '') + # `no-mistakes axi` with no subcommand: the home view, whose recent-runs + # table is the only public surface that carries run IDS (the top-level + # `runs` ledger has none). + printf '%s\n' "${FM_FAKE_AXI_HOME:-}" ;; esac ;; runs) @@ -220,6 +236,8 @@ arm_idle_record() { # reset_fakes() { FM_FAKE_AXI_STATUS="" FM_FAKE_AXI_STATUS_RUN="" + FM_FAKE_AXI_RUN_ID="" + FM_FAKE_AXI_HOME="" FM_FAKE_RUNS_LIST="" FM_FAKE_BUSY=0 FM_FAKE_BUSY_TEXT= @@ -234,7 +252,7 @@ reset_fakes() { FM_FAKE_HERDR_SHELL_PID=$$ FM_FAKE_CI_LOGS="" FM_FAKE_DAEMON_DOWN=0 - export FM_FAKE_AXI_STATUS FM_FAKE_AXI_STATUS_RUN FM_FAKE_RUNS_LIST FM_FAKE_BUSY FM_FAKE_BUSY_TEXT FM_FAKE_TMUX_MISSING FM_FAKE_TMUX_UNREADABLE + export FM_FAKE_AXI_STATUS FM_FAKE_AXI_STATUS_RUN FM_FAKE_AXI_RUN_ID FM_FAKE_AXI_HOME FM_FAKE_RUNS_LIST FM_FAKE_BUSY FM_FAKE_BUSY_TEXT FM_FAKE_TMUX_MISSING FM_FAKE_TMUX_UNREADABLE export FM_FAKE_HERDR_BUSY FM_FAKE_HERDR_MISSING FM_FAKE_HERDR_READ_FAIL FM_FAKE_HERDR_HUSK FM_FAKE_HERDR_AGENT_STATUS FM_FAKE_HERDR_PROCESS FM_FAKE_HERDR_SHELL_PID FM_FAKE_CI_LOGS export FM_FAKE_DAEMON_DOWN } @@ -365,6 +383,71 @@ steps[3]{step,status,findings,duration_ms}: EOF } +# The live sibling run's own detail, in the shape `no-mistakes axi status --run +# ` really returns (captured from run 01M2H391VCGZ6TX3ZKQKBHDBZ4 on the +# installed CLI v1.72.0): the run status word stays `running` - the ledger's own +# vocabulary has no gate words in it - the park shows up as `awaiting_agent`, +# the steps table is nested under `run:`, and the per-run query adds the +# top-level `gate:` block with the gate's findings table that the crew's own +# bare `axi status` omits. Carries the 2026-09-15 incident's numbers: parked +# 23m57s at review, two findings, one of them ask-user. +run_parked_live_sibling() { # + cat < + cat <... + local row + printf 'current_branch: fm/whatever\ndaemon: running\ncount: %d of %d total\n' "$#" "$#" + printf 'runs[%d]{id,branch,status,head,pr}:\n' "$#" + for row in "$@"; do printf ' %s\n' "$row"; done +} + run_passed() { # cat </dev/null + fm_write_meta "$d/state/feat-await.meta" "window=fm:fm-feat-await" "worktree=$d/wt" "kind=ship" + FM_FAKE_AXI_STATUS="$(run_parked_awaiting_only fm/feat-await)" + local out; out=$(run_crew_state "$d" feat-await) + assert_equals "state: parked · source: run-step · parked at ci: 3 finding(s)" "$out" \ + "a park visible only as a steps-table row must still name that gate" + assert_not_contains "$out" "state: working" "a run waiting on a human is not active work" + pass "an awaiting_agent run with no gate block reports its step row's gate" +} + test_ci_ready_done_log_beats_monitoring_run() { reset_fakes local d; d=$(new_case ci-ready) @@ -1243,6 +1343,461 @@ EOF pass "an unfetched live sibling outranks a terminal row at the worktree's exact commit" } +# The nutrifam-cerrar-allow-authenticated incident (2026-09-07): the live run +# REBASED the branch onto an advanced upstream, so its head is a new commit on +# a line of history the worktree HEAD is not an ancestor of, and the head rule +# rejects it in both directions. `axi status` answered with the previous run, +# which had died at this worktree's exact commit, so the terminal answer stood +# and a healthy task parked at its gate was reported failed - and the watcher +# turned that into a terminal-outcome wake for an outcome that never happened. +# A live row for this worktree's own branch is the present whether or not its +# rebased head still binds, once an older row on that branch binds the worktree. +test_rebased_live_run_outranks_terminal_row_at_worktree_head() { + reset_fakes + local d base_head rebased_head short_base short_rebased out + d=$(new_case rebased-live-run) + make_repo_on_branch "$d/wt" fm/feat-rebased + git -C "$d/wt" commit -q --allow-empty -m 'the work this crew submitted' + base_head=$(git -C "$d/wt" rev-parse HEAD) + # The live run's head: the branch replayed onto an advanced upstream, so it + # resolves in the task copy (the pipeline pushed it) but shares no ancestry + # with the worktree HEAD in either direction. + git -C "$d/wt" checkout -q --detach "$(git -C "$d/wt" rev-list --max-parents=0 HEAD)" + git -C "$d/wt" commit -q --allow-empty -m 'upstream advanced' + git -C "$d/wt" commit -q --allow-empty -m 'the pipeline replayed the branch onto it' + rebased_head=$(git -C "$d/wt" rev-parse HEAD) + git -C "$d/wt" checkout -q fm/feat-rebased + git -C "$d/wt" merge-base --is-ancestor "$base_head" "$rebased_head" \ + && fail "the rebased head must not descend from the worktree HEAD" + git -C "$d/wt" merge-base --is-ancestor "$rebased_head" "$base_head" \ + && fail "the worktree HEAD must not descend from the rebased head" + short_base=$(git -C "$d/wt" rev-parse --short=7 "$base_head") + short_rebased=$(git -C "$d/wt" rev-parse --short=7 "$rebased_head") + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/rebased.meta" "window=fm:fm-rebased" "worktree=$d/wt" "kind=ship" + # The dead run sits at this worktree's exact commit, so it binds and answers. + FM_FAKE_RUN_HEAD="$base_head" + FM_FAKE_AXI_STATUS="$(run_failed fm/feat-rebased)" + FM_FAKE_RUNS_LIST="$(cat </dev/null 2>&1 \ + && fail "the live run's head must not resolve in the task copy" + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/parkedlive.meta" "window=fm:fm-parkedlive" \ + "worktree=$d/wt" "kind=ship" + # Bare `axi status` binds the previous FAILED run by exact head equality. + FM_FAKE_RUN_HEAD="$base_head" + FM_FAKE_AXI_STATUS="$(run_failed fm/feat-parkedlive)" + FM_FAKE_RUNS_LIST="$(cat </dev/null + fm_write_meta "$d/state/wronghead.meta" "window=fm:fm-wronghead" \ + "worktree=$d/wt" "kind=ship" + FM_FAKE_RUN_HEAD="$base_head" + FM_FAKE_AXI_STATUS="$(run_failed fm/feat-wronghead)" + FM_FAKE_RUNS_LIST="$(cat </dev/null + fm_write_meta "$d/state/emptyhead.meta" "window=fm:fm-emptyhead" \ + "worktree=$d/wt" "kind=ship" + FM_FAKE_RUN_HEAD="$base_head" + FM_FAKE_AXI_STATUS="$(run_failed fm/feat-emptyhead)" + FM_FAKE_RUNS_LIST="$(cat </dev/null + fm_write_meta "$d/state/foreigndetail.meta" "window=fm:fm-foreigndetail" \ + "worktree=$d/wt" "kind=ship" + FM_FAKE_RUN_HEAD="$base_head" + FM_FAKE_AXI_STATUS="$(run_failed fm/feat-foreigndetail)" + FM_FAKE_RUNS_LIST="$(cat </dev/null + fm_write_meta "$d/state/terminaldetail.meta" "window=fm:fm-terminaldetail" \ + "worktree=$d/wt" "kind=ship" + FM_FAKE_RUN_HEAD="$base_head" + FM_FAKE_AXI_STATUS="$(run_failed fm/feat-terminaldetail)" + FM_FAKE_RUNS_LIST="$(cat </dev/null + fm_write_meta "$d/state/otherbranch.meta" "window=fm:fm-otherbranch" \ + "worktree=$d/wt" "kind=ship" + FM_FAKE_AXI_STATUS="$(run_running fm/other-crew)" + FM_FAKE_RUNS_LIST="$(cat </dev/null + fm_write_meta "$d/state/statustable.meta" "window=fm:fm-statustable" \ + "worktree=$d/wt" "kind=ship" + FM_FAKE_AXI_STATUS="$(cat </dev/null + fm_write_meta "$d/state/rebasednoanchor.meta" "window=fm:fm-rebasednoanchor" \ + "worktree=$d/wt" "kind=ship" "harness=claude" + FM_FAKE_AXI_STATUS="$(run_running fm/other-crew)" + FM_FAKE_RUNS_LIST="$(cat </dev/null + fm_write_meta "$d/state/ancestorlive.meta" "window=fm:fm-ancestorlive" \ + "worktree=$d/wt" "kind=ship" "harness=claude" + FM_FAKE_AXI_STATUS="$(run_running fm/other-crew)" + FM_FAKE_RUNS_LIST="$(cat </dev/null + fm_write_meta "$d/state/divergedterminal.meta" "window=fm:fm-divergedterminal" \ + "worktree=$d/wt" "kind=ship" "harness=claude" + FM_FAKE_AXI_STATUS="$(run_running fm/other-crew)" + FM_FAKE_RUNS_LIST="$(cat < + cat > "$1/no-mistakes" < local home=$TMP_ROOT/$1 mkdir -p "$home/state" "$home/data" "$home/projects" "$home/config" @@ -476,6 +531,51 @@ test_event_hints_follow_reconciled_current_state() { pass "snapshot event hints follow reconciled current state" } +# The 2026-09-15 false-active report, end to end through the classifier the +# captain's fleet view is built from: a crew whose live run is PARKED at an +# unanswered gate, and whose run bare `axi status` cannot bind to the worktree, +# must land in the waiting list, never in active work. Only `parked`, `paused` +# and `blocked` reach $holds_all; `working` counts as work in progress, which is +# how a crew sitting on an ask-user gate could stay invisible indefinitely. +test_parked_live_run_is_a_hold_not_active_work() { + local home fakebin out repo base short + home=$(make_home parked-live-run) + repo="$home/projects/parkedlive" + mkdir -p "$repo" + fm_git_identity fmtest fmtest@example.invalid + git -C "$repo" init -q + git -C "$repo" commit -q --allow-empty -m init + git -C "$repo" checkout -q -b fm/feat-parkedlive + base=$(git -C "$repo" rev-parse HEAD) + short=$(git -C "$repo" rev-parse --short=7 "$base") + cat > "$home/data/backlog.md" <<'EOF' +## In flight +- [ ] parkedlive - Crew parked at an unanswered gate (repo: alpha) (kind: ship) (since 2026-09-15) + +## Queued + +## Done +EOF + fm_write_meta "$home/state/parkedlive.meta" \ + "window=firstmate:fm-parkedlive" \ + "worktree=$repo" \ + "project=alpha" \ + "harness=claude" \ + "kind=ship" \ + "mode=no-mistakes" + fakebin=$(make_fakebin "$home") + write_parked_run_fake "$fakebin" fm/feat-parkedlive "$short" + out=$(PATH="$fakebin:$PATH" FM_HOME="$home" "$SNAPSHOT" --secondmate-home-summary) + printf '%s' "$out" | jq -e ' + (.holds | any(.id == "parkedlive")) + and (.holds[] | select(.id == "parkedlive") | .reason | contains("parked at review")) + and (.holds[] | select(.id == "parkedlive") | .source == "child-state") + and .active_children == [] + and .counts.active_children == 0 + ' >/dev/null || fail "a crew parked at an unanswered gate was not classified as a wait: $out" + pass "a live run parked at its gate is a hold, not active work" +} + test_scout_reports_include_teardown_reports() { local home out home=$(make_home teardown-reports) @@ -1059,6 +1159,7 @@ test_open_decision_transfers_to_captain_hold test_open_decision_clears_on_keyed_resolution test_completed_scout_report_is_pointer_not_pending test_parked_scout_decision_stays_pending +test_parked_live_run_is_a_hold_not_active_work test_scout_reports_include_teardown_reports test_backlog_tasks_axi_forms_and_overrides test_view_renders_snapshot