diff --git a/AGENTS.md b/AGENTS.md index 0a83bf0daed..c22bfbdba23 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -89,6 +89,7 @@ projects/ cloned repos; gitignored; read-only except under hard rule state/ volatile runtime signals; gitignored .status appended by crewmates: ": " wake-event lines, not current-state truth .turn-ended task completion signal; harness-adapters routes each producer contract to its authoritative implementation + .run-step last observed no-mistakes run step; private cache owned by bin/fm-crew-state.sh and removed by teardown .agy-trust agy trust cleanup marker; exact lifecycle lives in bin/fm-agy-trust-lib.sh .grok-turnend-token firstmate-owned grok hook registry token for the task; removed by teardown .kimi-turnend-token firstmate-owned Kimi hook registry token for the task; removed by teardown @@ -326,7 +327,7 @@ Send the same worker one exact decision naming the decision key, step, action, a Require the matching `resolved` event, forbid `--yes`, and require the worker to process every synchronous return until completion or a genuinely new escalation. Resume fleet supervision immediately after the decision lands. -Judge validation by the current-code-matched run step through `bin/fm-crew-state.sh`, not by shell liveness or the last status event. +Judge validation by the reconciled run-step verdict from `bin/fm-crew-state.sh`, not by shell liveness or the last status event. Running, fixing, or CI states remain working; parked approval or fix-review states require the worker to follow the active gate help; passed or checks-passed is done; failed or cancelled is failed. A worker hand-editing, committing, aborting, or restarting during an active validation run duplicates pipeline ownership outside the supersession sequence above; steer it back to the gate response flow. The worker reports the PR when CI first becomes green rather than waiting for merge monitoring to finish. diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index d80840f6a13..468b77e0408 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -318,7 +318,7 @@ signal_reason_is_actionable() { # ... # Classify WHY an idle/stale crew MIGHT be safely absorbed instead of surfaced, # from bin/fm-crew-state.sh's one authoritative current-state line # ("state: · source: · "). Prints exactly one token: -# working - an actively-running no-mistakes step (running/fixing/ci) or a busy +# working - a current no-mistakes step, its bounded degraded replay, or a busy # pane; the crew is legitimately mid-work on a static-looking pane # (e.g. waiting on CI); # paused - the crew's authoritative current state is a declared external-wait @@ -340,7 +340,9 @@ crew_absorb_class() { # if [ "$state" = paused ]; then printf 'paused'; return; fi if [ "$state" = working ]; then src=${line#*source: }; src=${src%% *} - case "$src" in run-step|pane) printf 'working'; return ;; esac + # A working run-step-degraded verdict is bounded positive evidence whose + # replay and safety rules are owned by fm-crew-state.sh. + case "$src" in run-step|run-step-degraded|pane) printf 'working'; return ;; esac fi printf 'none' } diff --git a/bin/fm-crew-state.sh b/bin/fm-crew-state.sh index 30fc7b7236e..735672c42af 100755 --- a/bin/fm-crew-state.sh +++ b/bin/fm-crew-state.sh @@ -7,27 +7,34 @@ # last EVENT, not the current STATE. After firstmate resolves a needs-decision # or blocked and the crew resumes (responds to the gate, the pipeline fixes, it # re-validates), the log's last line stays stale. This helper never infers the -# current state from a tail of the log: it reads the authoritative source (a -# no-mistakes run-step attributed to this crew's branch and current code -# identity, else the pane busy-signature) and reconciles the possibly-stale log -# against it. +# current state from a tail of the log: it reconciles attributable no-mistakes run +# evidence, the bounded record used only after a lookup failure, and pane busy +# state before consulting the possibly-stale log. # -# The determinism lives entirely here - only run-step / pane / log reads plus -# fixed mapping logic, no heuristics and no LLM. Output is one stable, parseable, -# token-tight line firstmate can read every heartbeat: +# The deterministic reconciliation lives entirely here - bounded run-step / pane / +# log reads, one run-step cache update, and fixed mapping logic, with no heuristics +# and no LLM. Output is one stable, parseable, token-tight line firstmate can read +# every heartbeat: # -# state: · source: · +# state: · source: · # # Logic, in order: # 1. Resolve worktree + backend target + kind from state/.meta. -# 2. Matching no-mistakes run for this crew's branch AND current code identity, +# 2. Matching no-mistakes run for this crew's branch AND attributable code identity, # active or terminal (from `axi status`, or the coarse `no-mistakes runs` # fallback)? Branch name alone is not enough: a historical run on a reused # branch whose head was rewritten or diverged must not be attributed. # A run matches when its head equals the worktree HEAD, or the worktree HEAD # is an ancestor of the run head (pipeline fix commits advanced the run on -# the same line of history). Local work that advanced past the run head, or -# diverged from it, invalidates attribution. +# the same line of history). It also matches in two cases the pipeline's own +# commits create, both narrowed to the run that is demonstrably current: +# an ACTIVELY-EXECUTING run whose head the tip has advanced past (it +# authored that advance), and a LIVE run answered for this branch whose head +# is not an object in this worktree at all (the pipeline owns the branch and +# commits in a copy this worktree has never fetched). Local work that +# advanced past a parked or terminal run head, a rewritten or diverged tip, +# and an unresolvable sha read out of the historical runs listing all still +# invalidate attribution. nm_head_attributable owns the exact rule. # The run-step is AUTHORITATIVE: running/fixing -> working, ci -> working, # awaiting_approval/fix_review -> parked (with gate findings), terminal # passed/checks-passed -> done, failed/cancelled -> failed. EXCEPT: while @@ -39,16 +46,26 @@ # the run-step shows the run moved on, the log is deterministically stale and # is flagged superseded. A genuinely parked run plus a needs-decision log # agree, and are reported as parked. -# 4. No run for this crew (pre-validation, or kind=scout): fall back to the -# recorded backend's pane busy state, then the status log's last line only -# when its verb maps to a recognized run-state. Decision-only events such as -# `resolved` never become current state or detail. -# 5. Missing meta or torn-down worktree: report unknown · none. If no run is +# 4. A run lookup that cannot complete may replay this crew's recent observed +# step as `run-step-degraded`, but only after endpoint liveness and an exact +# busy verdict and only inside the configured age bound. +# 5. A completed lookup with no run for this crew (pre-validation, or kind=scout) +# falls back to the recorded backend's pane busy state, then the status log's +# last line only when its verb maps to a recognized run-state. Decision-only +# events such as `resolved` never become current state or detail. +# 6. Missing meta or torn-down worktree: report unknown · none. If no run is # attributed to this crew, a dead endpoint also reports unknown · none rather # than trusting a stale status log. # -# Read-only and side-effect free. Always exits 0 on a successful read regardless -# of state; exit 2 only on a usage error (no id). +# A run LOOKUP FAILURE is not a run ABSENCE. The bounded no-mistakes call +# propagates timeout, execution, and no-bounding-mechanism failures so only a +# failed lookup can replay the last known run step; a completed lookup that found +# no run falls through to the pane and log sources. +# +# Writes exactly one thing: state/.run-step, the last known run-step record +# that makes that degrade possible (runstep_record_write below is the only +# writer). Every other read is side-effect free. Always exits 0 on a successful +# read regardless of state; exit 2 only on a usage error (no id). set -u SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -78,6 +95,16 @@ case "$NM_TIMEOUT" in ''|*[!0-9]*) NM_TIMEOUT=10 ;; esac # history every call. FM_CREW_STATE_RUNS_LIMIT=${FM_CREW_STATE_RUNS_LIMIT:-200} case "$FM_CREW_STATE_RUNS_LIMIT" in ''|*[!0-9]*) FM_CREW_STATE_RUNS_LIMIT=200 ;; esac +# How long a recorded run-step stays usable as the degraded answer after the run +# lookup starts failing. This is the bound that keeps the degrade from becoming a +# worse bug than the one it fixes: a permanently broken no-mistakes daemon would +# otherwise let every idle crew claim it is still validating forever, and a real +# wedge would never surface again. Past this age the record is ignored and the +# crew falls through to the pane/log sources exactly as it does today. 0 disables +# the degrade entirely, restoring the strict "cannot re-confirm it, do not claim +# it" reading for a home that wants it. +FM_CREW_STATE_DEGRADED_MAX_AGE=${FM_CREW_STATE_DEGRADED_MAX_AGE:-900} +case "$FM_CREW_STATE_DEGRADED_MAX_AGE" in ''|*[!0-9]*) FM_CREW_STATE_DEGRADED_MAX_AGE=900 ;; esac SEP=' · ' # Emit the one canonical line and exit 0. Detail is optional. @@ -166,6 +193,57 @@ crew_busy_verdict() { # fm_busy_classify "$TASK_BACKEND" "$1" "$HARNESS" "$ID" "$STATE" "$tail40" } +# --- last known run-step record --------------------------------------------- +# state/.run-step: one line, "\t\t", atomically +# replaced on every successful run-derived verdict and read back ONLY when a +# later lookup could not complete. It is what lets a lookup FAILURE answer +# "still validating, lookup unavailable" instead of "unknown", without ever +# inventing evidence: nothing is degraded for a crew that was never seen +# validating in the first place, so a crew that genuinely stopped before any run +# has no record to fall back on and still surfaces as a wedge suspect. +RUNSTEP_RECORD="$STATE/$ID.run-step" + +# Record a run-derived verdict. Never fails the read: a state dir that is +# read-only or already torn down just leaves the previous record in place. +runstep_record_write() { # + local tmp flat + case "$1" in working|parked|done|failed) ;; *) return 0 ;; esac + [ -d "$STATE" ] || return 0 + flat=$(printf '%s' "${2:-}" | tr '\t\n' ' ') + tmp="$RUNSTEP_RECORD.$$.tmp" + if printf '%s\t%s\t%s\n' "$(date +%s)" "$1" "$flat" > "$tmp" 2>/dev/null; then + mv -f "$tmp" "$RUNSTEP_RECORD" 2>/dev/null || rm -f "$tmp" 2>/dev/null + else + rm -f "$tmp" 2>/dev/null + fi + return 0 +} + +runstep_record_clear() { + rm -f "$RUNSTEP_RECORD" 2>/dev/null || true +} + +runstep_record_emit() { # + runstep_record_write "$1" "${3:-}" + emit "$1" "$2" "${3:-}" +} + +# Print "\t\t" for a record still inside the +# freshness bound; return 1 for a missing, malformed, or expired record. +runstep_record_read() { + local ts st detail now age + [ -f "$RUNSTEP_RECORD" ] || return 1 + IFS=$'\t' read -r ts st detail < "$RUNSTEP_RECORD" 2>/dev/null || return 1 + case "${ts:-}" in ''|*[!0-9]*) return 1 ;; esac + case "${st:-}" in working|parked|done|failed) ;; *) return 1 ;; esac + now=$(date +%s) + case "$now" in ''|*[!0-9]*) return 1 ;; esac + age=$(( now - ts )) + [ "$age" -ge 0 ] || return 1 + [ "$age" -lt "$FM_CREW_STATE_DEGRADED_MAX_AGE" ] || return 1 + printf '%s\t%s\t%s' "$st" "${detail:-}" "$age" +} + # --- no-mistakes run lookup (authoritative when a run matches this branch) -- trim() { @@ -183,7 +261,12 @@ strip_quotes() { trim "$s" } -# Bounded no-mistakes call in the worktree; stdout only, never fails the script. +# Bounded no-mistakes call in the worktree; stdout only, never fails the script +# (there is no `set -e`). The EXIT STATUS is deliberately propagated rather than +# swallowed: 124 from a timeout, or any other non-zero, is what tells the caller +# the lookup could not COMPLETE, which must never be read as "this branch has no +# run". With no bounding mechanism available at all the call is not made, and 127 +# reports that same inability rather than a silent empty answer. HAVE_TIMEOUT=none if command -v timeout >/dev/null 2>&1; then HAVE_TIMEOUT=timeout elif command -v gtimeout >/dev/null 2>&1; then HAVE_TIMEOUT=gtimeout @@ -191,10 +274,10 @@ elif command -v perl >/dev/null 2>&1; then HAVE_TIMEOUT=perl fi nm_run() { # case "$HAVE_TIMEOUT" in - timeout) ( cd "$WT" && timeout "$NM_TIMEOUT" no-mistakes "$@" ) 2>/dev/null || true ;; - gtimeout) ( cd "$WT" && gtimeout "$NM_TIMEOUT" no-mistakes "$@" ) 2>/dev/null || true ;; - perl) ( cd "$WT" && perl -e 'my $t = shift; my $pid = fork; die "fork failed" unless defined $pid; if (!$pid) { setpgrp(0, 0); exec @ARGV } local $SIG{ALRM} = sub { kill "TERM", -$pid; select undef, undef, undef, 0.2; kill "KILL", -$pid; exit 124 }; alarm $t; waitpid $pid, 0; exit($? >> 8)' "$NM_TIMEOUT" no-mistakes "$@" ) 2>/dev/null || true ;; - *) true ;; + timeout) ( cd "$WT" && timeout "$NM_TIMEOUT" no-mistakes "$@" ) 2>/dev/null ;; + gtimeout) ( cd "$WT" && gtimeout "$NM_TIMEOUT" no-mistakes "$@" ) 2>/dev/null ;; + perl) ( cd "$WT" && perl -e 'my $t = shift; my $pid = fork; die "fork failed" unless defined $pid; if (!$pid) { setpgrp(0, 0); exec @ARGV } local $SIG{ALRM} = sub { kill "TERM", -$pid; select undef, undef, undef, 0.2; kill "KILL", -$pid; exit 124 }; alarm $t; waitpid $pid, 0; exit($? >> 8)' "$NM_TIMEOUT" no-mistakes "$@" ) 2>/dev/null ;; + *) return 127 ;; esac } @@ -352,10 +435,14 @@ nm_ci_checks_state() { # is exact) - but branch + coarse status is exactly what this predicate needs: # is a run for THIS branch active right now. Echoes the first (most recent) # matching row's status word (running/completed/cancelled/failed), or empty -# when the branch has no run within FM_CREW_STATE_RUNS_LIMIT rows. -nm_runs_status_for_branch() { # - local branch=$1 out row st rest br sha - out=$(nm_run runs --limit "$FM_CREW_STATE_RUNS_LIMIT") +# when the branch has no attributable run in the listing. +# +# A PURE PARSER over a listing the caller already captured. The call itself is +# made by the caller so a listing that could not be fetched is classified as a +# lookup failure there; parsing an empty string here would otherwise report the +# same "no run for this branch" as a listing that genuinely lacks the branch. +nm_runs_status_for_branch() { # + local branch=$1 out=${2:-} row st rest br sha [ -n "$out" ] || return 0 while IFS= read -r row; do row=$(trim "$row") @@ -368,11 +455,7 @@ nm_runs_status_for_branch() { # rest=$(trim "$rest") sha=${rest%% *} if [ "$br" = "$branch" ]; then - # Same code-identity rule as axi status: skip a same-branch row whose - # short-sha does not match this worktree (rewritten or advanced tip). - if ! nm_coarse_head_matches_worktree "$sha"; then - continue - fi + nm_head_attributable "$sha" 0 0 || continue printf '%s' "$st" return 0 fi @@ -384,41 +467,117 @@ nm_runs_status_for_branch() { # # scratch worktree); with no branch there is no run to attribute to this crew. CREW_BRANCH=$(git -C "$WT" symbolic-ref --quiet --short HEAD 2>/dev/null || true) -# 0 if the active axi-status run's head field matches this worktree's code -# identity. Branch match is a precondition (caller). Rules: -# - missing/empty head field: cannot bind; reject the run -# - equal commits (short or full SHA): match -# - worktree HEAD is an ancestor of run head: match (pipeline fix commits on -# the same history advanced the run tip) -# - run head is a strict ancestor of worktree HEAD: no match (local work -# advanced outside the run) -# - diverged / run head not in this worktree: no match (rewritten branch tip) -nm_run_head_matches_worktree() { - local run_head local_full run_full - run_head=$(strip_quotes "$(nm_field head)") - [ -n "$run_head" ] || return 1 - local_full=$(git -C "$WT" rev-parse HEAD 2>/dev/null) || return 1 - run_full=$(git -C "$WT" rev-parse --verify "${run_head}^{commit}" 2>/dev/null) || return 1 - [ "$run_full" = "$local_full" ] && return 0 +# How a run's recorded head relates to this worktree's HEAD. One owner for +# the commit-relationship test both the `axi status` head field and the coarse +# runs-list short sha are judged by. Prints exactly one token: +# equal the run head IS the worktree HEAD +# run-ahead worktree HEAD is an ancestor of the run head - pipeline fix +# commits advanced the run tip past what this worktree has read +# run-behind the run head is a strict ancestor of the worktree HEAD - the tip +# advanced past the sha this run recorded +# unresolved the sha is not an object in this worktree at all, so no +# relationship can be computed (the pipeline is committing in a +# copy this worktree has never fetched from) +# missing no sha to judge, or this worktree has no readable HEAD +# diverged resolvable but on neither side of the worktree HEAD - a +# rewritten branch tip +nm_head_relation() { # + local run_head=${1:-} local_full run_full + [ -n "$run_head" ] || { printf 'missing'; return; } + local_full=$(git -C "$WT" rev-parse HEAD 2>/dev/null) || { printf 'missing'; return; } + run_full=$(git -C "$WT" rev-parse --verify "${run_head}^{commit}" 2>/dev/null) || { printf 'unresolved'; return; } + if [ "$run_full" = "$local_full" ]; then printf 'equal'; return; fi if git -C "$WT" merge-base --is-ancestor "$local_full" "$run_full" 2>/dev/null; then - return 0 + printf 'run-ahead'; return fi - return 1 -} - -# Coarse runs-list rows are " ...". 0 if the short -# sha for this branch row matches the worktree head under the same rules as -# nm_run_head_matches_worktree (equal, or local is ancestor of run tip). -nm_coarse_head_matches_worktree() { # - local run_head=$1 local_full run_full - [ -n "$run_head" ] || return 1 - local_full=$(git -C "$WT" rev-parse HEAD 2>/dev/null) || return 1 - run_full=$(git -C "$WT" rev-parse --verify "${run_head}^{commit}" 2>/dev/null) || return 1 - [ "$run_full" = "$local_full" ] && return 0 - if git -C "$WT" merge-base --is-ancestor "$local_full" "$run_full" 2>/dev/null; then - return 0 + if git -C "$WT" merge-base --is-ancestor "$run_full" "$local_full" 2>/dev/null; then + printf 'run-behind'; return + fi + printf 'diverged' +} + +# 0 when a run recorded at may be attributed to this worktree's current +# code. Branch match is a precondition (caller). is 1 only while the +# run sits in an actively-executing step - the states in which the pipeline +# commits its OWN fixes. is 1 only for an answer the CLI gave for +# THIS worktree's current branch, never for a row read out of the historical +# runs listing. +# +# `run-behind` is the fix-round case. When a review finding is answered +# `--action fix`, the pipeline commits that fix and the branch tip advances past +# the sha the run recorded; rejecting the row outright made the run that was +# CURRENTLY authoring those commits stop matching its own worktree, and the crew +# read as unknown for the rest of the fix round. An actively-executing run is the +# author of that advance and keeps attribution. A PARKED run is by definition +# waiting on a response and commits nothing, so a tip that advanced past it is +# local work outside the run and must still invalidate - as must a terminal run. +# +# `unresolved` is the pipeline-owned case, and is NOT the same evidence as a +# rewritten tip. While the pipeline owns the branch it commits in its own copy, +# so the head it reports is simply an object this worktree has never fetched; +# refusing it made a crew parked at a live fix_review gate read as having no run +# at all. Only a LIVE run answered for THIS branch earns that benefit: the +# historical runs listing has no notion of "current", so an unresolvable sha +# there stays rejected, and a terminal run's unseen head is evidence of nothing. +# `missing` and `diverged` are always rejected - an absent sha cannot bind, and a +# resolvable sha on neither side of HEAD is a genuinely rewritten branch. +nm_head_attributable() { # + case "$(nm_head_relation "$1")" in + equal|run-ahead) return 0 ;; + run-behind) [ "${2:-0}" = 1 ] && return 0; return 1 ;; + unresolved) [ "${3:-0}" = 1 ] && return 0; return 1 ;; + *) return 1 ;; + esac +} + +# 1 when the `axi status` run in $RUN_OUT is in an actively-executing step and so +# able to author pipeline fix commits; 0 for a parked, terminal, or unrecognized +# run. A run parked at a gate reports a plain `running` status in some shapes, so +# the gate markers are checked before the status word. +nm_run_is_authoring() { + local outcome status + outcome=$(strip_quotes "$(nm_field outcome)") + [ -z "$outcome" ] || { printf '0'; return; } + if printf '%s\n' "$RUN_OUT" | grep -Eq '^[[:space:]]*awaiting_agent:'; then + printf '0'; return fi - return 1 + if nm_has_gate; then printf '0'; return; fi + status=$(strip_quotes "$(nm_field status)") + case "$status" in + awaiting_approval|fix_review) printf '0' ;; + running|fixing|ci) printf '1' ;; + *) printf '0' ;; + esac +} + +# 1 when the `axi status` run in $RUN_OUT has not reached a terminal result, so +# it is still this branch's current run whether it is executing or parked at a +# gate. Broader than nm_run_is_authoring on purpose: a run parked at fix_review +# commits nothing right now but is emphatically still live. +nm_run_is_live() { + local outcome status + outcome=$(strip_quotes "$(nm_field outcome)") + [ -z "$outcome" ] || { printf '0'; return; } + status=$(strip_quotes "$(nm_field status)") + case "$status" in completed|failed|cancelled) printf '0' ;; *) printf '1' ;; esac +} + +# 0 if the axi-status run's head field is attributable to this worktree. The +# caller has already established this answer is for the crew's own branch, which +# is what makes it branch-scoped for the pipeline-owned rule above. +nm_run_head_matches_worktree() { + nm_head_attributable "$(strip_quotes "$(nm_field head)")" \ + "$(nm_run_is_authoring)" "$(nm_run_is_live)" +} + +nm_run_invalidates_record() { + local relation + [ "$(nm_run_is_live)" = 0 ] && return 0 + relation=$(nm_head_relation "$(strip_quotes "$(nm_field head)")") + case "$relation" in + run-behind|diverged) return 0 ;; + *) return 1 ;; + esac } HAVE_RUN=0 @@ -428,31 +587,63 @@ HAVE_RUN=0 # run-step block below skips the TOON field parsing entirely for this crew. RUN_SOURCE=full COARSE_STATUS="" +# LOOKUP_FAILED=1 means a bounded no-mistakes call could not COMPLETE - it timed +# out under a saturated daemon, errored, or could not be bounded at all. That is +# emphatically NOT the same as a completed lookup that found no run for this +# branch, and only a failure may degrade to the recorded run-step below. A +# genuine absence still falls through to the pane and status-log sources. +LOOKUP_FAILED=0 +LOOKUP_COMPLETED=0 # Scouts and secondmates never drive a no-mistakes validation of their own # worktree, so skip the lookup for them and read state from pane/log directly. if [ "$KIND" = ship ] && [ -n "$CREW_BRANCH" ] && command -v no-mistakes >/dev/null 2>&1; then RUN_OUT=$(nm_run axi status) - if [ -n "$RUN_OUT" ]; then + nm_rc=$? + # Empty stdout is a failure, not an absence: `axi status` answers a branch with + # no run of its own with some OTHER branch's run as informational display (the + # cross-branch case the coarse fallback below exists for), so it has no + # "nothing to report" empty answer to confuse this with. + if [ "$nm_rc" != 0 ] || [ -z "$RUN_OUT" ]; then + RUN_OUT="" + LOOKUP_FAILED=1 + else run_branch=$(strip_quotes "$(nm_field branch)") - if [ -n "$run_branch" ] && [ "$run_branch" = "$CREW_BRANCH" ] && nm_run_head_matches_worktree; then - HAVE_RUN=1 - else + if [ -n "$run_branch" ] && [ "$run_branch" = "$CREW_BRANCH" ]; then + if nm_run_head_matches_worktree; then + HAVE_RUN=1 + LOOKUP_COMPLETED=1 + elif nm_run_invalidates_record; then + runstep_record_clear + fi + fi + if [ "$HAVE_RUN" = 0 ]; then # The active-or-most-recent run is for another branch, or same branch with # a rewritten/diverged head (the CLI is alive and answered; only the # attribution missed) - try the coarse fallback. - # Deliberately nested inside `[ -n "$RUN_OUT" ]`: an empty/timed-out + # Deliberately reached only when the primary call ANSWERED: a timed-out # primary call means the CLI itself did not respond, so retrying it # immediately with a second bounded call would just double the wait # for no better answer. - COARSE_STATUS=$(nm_runs_status_for_branch "$CREW_BRANCH") - if [ -n "$COARSE_STATUS" ]; then - HAVE_RUN=1 - RUN_SOURCE=coarse + runs_out=$(nm_run runs --limit "$FM_CREW_STATE_RUNS_LIMIT") + runs_rc=$? + if [ "$runs_rc" != 0 ] || [ -z "$runs_out" ]; then + LOOKUP_FAILED=1 + else + LOOKUP_COMPLETED=1 + COARSE_STATUS=$(nm_runs_status_for_branch "$CREW_BRANCH" "$runs_out") + if [ -n "$COARSE_STATUS" ]; then + HAVE_RUN=1 + RUN_SOURCE=coarse + fi fi fi fi fi +if [ "$LOOKUP_COMPLETED" = 1 ] && [ "$HAVE_RUN" = 0 ]; then + runstep_record_clear +fi + # --- run-step authoritative path ------------------------------------------- if [ "$HAVE_RUN" = 1 ]; then @@ -538,7 +729,7 @@ if [ "$HAVE_RUN" = 1 ]; then if [ "$RUN_STATE" = working ] && log_reports_ci_ready; then if [ "$RUN_SOURCE" = coarse ]; then - emit "done" status-log "$(status_line_note "$LOG_LINE")${SEP}run still monitoring PR" + runstep_record_emit "done" status-log "$(status_line_note "$LOG_LINE")${SEP}run still monitoring PR" fi [ -n "$CI_STEP_STATUS" ] || CI_STEP_STATUS=$(nm_effective_ci_step_status) if [ "$RUN_STATUS" = fixing ]; then @@ -549,7 +740,7 @@ if [ "$HAVE_RUN" = 1 ]; then CI_LOG_STATE=not-ready fi if [ "$CI_LOG_STATE" != not-ready ]; then - emit "done" status-log "$(status_line_note "$LOG_LINE")${SEP}run still monitoring PR" + runstep_record_emit "done" status-log "$(status_line_note "$LOG_LINE")${SEP}run still monitoring PR" fi fi @@ -568,14 +759,20 @@ if [ "$HAVE_RUN" = 1 ]; then ;; esac - emit "$RUN_STATE" run-step "$RUN_DETAIL" + # Remember this verdict so a later lookup that cannot complete degrades to it + # instead of collapsing to unknown. Recorded from the authoritative run-step + # path only, so nothing but a genuinely observed run is ever replayed. + runstep_record_emit "$RUN_STATE" run-step "$RUN_DETAIL" fi # --- fallback: no run attributed to this crew ------------------------------ -# The run-step path above already handled any crew with a run, regardless of pane -# liveness, so a finished-but-pane-closed crew never reaches here. Down here there -# is no run to consult, so a dead/unreadable target means the crew is gone: report -# unknown rather than trusting a possibly-stale status log as the current state. +# The run-step path above already handled any crew with an attributed run, +# regardless of pane liveness, so a finished-but-pane-closed crew never reaches +# here. Down here either the lookup completed and found no run, or it could not +# complete at all; in both cases there is no live run to consult, so a +# dead/unreadable target means the crew is gone: report unknown rather than +# trusting a possibly-stale status log - or a remembered run-step - as the +# current state. [ -n "$BACKEND_TARGET" ] || emit unknown none "no backend target recorded" pane_readable "$BACKEND_TARGET" || emit unknown none "backend target gone: $BACKEND_TARGET" @@ -583,16 +780,46 @@ pane_readable "$BACKEND_TARGET" || emit unknown none "backend target gone: $BACK # state is not meaningful for them; read their state from the status log only. # Only an exact busy verdict reports working here, and only an exact idle # verdict permits the status-log fallback below. Missing, malformed, stale, or -# unverified semantic state remains unknown. +# unverified semantic state remains unknown - deferred rather than emitted here +# only so the degraded run-step below can answer first; it still outranks the +# status-log fallback exactly as before. +BUSY_STATE="" +BUSY_VERDICT="" if [ "$KIND" != secondmate ]; then BUSY_VERDICT=$(crew_busy_verdict "$BACKEND_TARGET") case "${BUSY_VERDICT%% *}" in busy) emit working pane "harness busy (${BUSY_VERDICT#* })" ;; - idle) ;; - *) emit unknown pane "harness state unavailable ($BUSY_VERDICT)" ;; + idle) BUSY_STATE=idle ;; + *) BUSY_STATE=unknown ;; esac fi +# The run lookup could not complete, and this crew has a recent run-step on +# record: report that last known step, degraded, rather than unknown. Bounded by +# FM_CREW_STATE_DEGRADED_MAX_AGE so a permanently unreachable daemon stops +# absorbing wedge suspicion instead of hiding it forever, and never reached at +# all on a completed lookup that simply found no run. +# +# Placement is the safety property. It sits BELOW the endpoint checks and the +# exact busy verdict, so live positive evidence - a gone endpoint proving the +# crew stopped, or a busy harness proving it is working right now - always +# outranks a remembered step. It sits ABOVE the unreadable-harness and +# status-log fallbacks, which is the whole point: an observed run-step, even one +# that could not be re-confirmed this poll, is better current-state evidence +# than an append-only event log. +if [ "$LOOKUP_FAILED" = 1 ]; then + if DEGRADED=$(runstep_record_read); then + IFS=$'\t' read -r deg_state deg_detail deg_age <<< "$DEGRADED" + deg_line="run lookup unavailable" + [ -n "$deg_detail" ] && deg_line="$deg_detail${SEP}$deg_line" + emit "$deg_state" run-step-degraded "$deg_line (last known ${deg_age}s ago)" + fi +fi + +if [ "$BUSY_STATE" = unknown ]; then + emit unknown pane "harness state unavailable ($BUSY_VERDICT)" +fi + # Fall back to the status log's last line, but ONLY when its verb maps to a real # run-state. A decision-closing event - resolved: (fm-classify-lib.sh's # FM_CLASSIFY_RESOLVE_VERB), and any future decision-only sibling - is NOT a state: diff --git a/bin/fm-fleet-snapshot.sh b/bin/fm-fleet-snapshot.sh index 3dde167d483..8ff49b20f66 100755 --- a/bin/fm-fleet-snapshot.sh +++ b/bin/fm-fleet-snapshot.sh @@ -478,7 +478,10 @@ task_json_lines() { # multiplex many concerns onto one stream, so activity on one concern must # never clear another concern's keyed decision. A parked/blocked state, or a # non-authoritative status-log/none read on a still-live task, keeps the fold's - # open decision surfacing. + # open decision surfacing. `run-step-degraded` is deliberately absent from the + # live-activity sources: it is a remembered step the reader could not + # re-confirm, which is enough to keep a crew provably working for wedge triage + # but never enough to clear a captain decision. open_decisions_tsv=$(status_open_decisions "$status_log") if [ "$kind" != secondmate ] && \ { { { [ "$current_source" = run-step ] || [ "$current_source" = pane ]; } \ diff --git a/bin/fm-teardown.sh b/bin/fm-teardown.sh index ecfa8461e2a..7a101b39bc7 100755 --- a/bin/fm-teardown.sh +++ b/bin/fm-teardown.sh @@ -1676,6 +1676,7 @@ cleanup_firstmate_home_children() { retire_busy_state "$sub_state" "$child_id" "$child_busy_gen" || return 1 rm -f "$sub_state/$child_id.status" "$sub_state/$child_id.turn-ended" \ "$sub_state/$child_id.meta" "$sub_state/$child_id.pi-ext.ts" \ + "$sub_state/$child_id.run-step" \ "$sub_state/$child_id.grok-turnend-token" "$sub_state/$child_id.kimi-turnend-token" done } @@ -1934,7 +1935,8 @@ fi remove_pr_poll_artifacts "$STATE" "$ID" || exit 1 retire_busy_state "$STATE" "$ID" "$BUSY_GEN" || exit 1 rm -f "$STATE/$ID.status" "$STATE/$ID.turn-ended" "$STATE/$ID.meta" \ - "$STATE/$ID.pi-ext.ts" "$STATE/$ID.grok-turnend-token" \ + "$STATE/$ID.pi-ext.ts" "$STATE/$ID.run-step" \ + "$STATE/$ID.grok-turnend-token" \ "$STATE/$ID.kimi-turnend-token" if [ "$KIND" != scout ] && [ "$KIND" != secondmate ] && [ "$MODE" != local-only ]; then "$FM_ROOT/bin/fm-fleet-sync.sh" "$PROJ" || true diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 5df7f336d26..549ec680f07 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -4,14 +4,14 @@ # and keeps blocking; it queues and exits only for actionable wakes. # The no-verb signal and first-sighting stale paths are # absorb-only-when-provably-working: a wake is absorbed only when the crew shows -# POSITIVE evidence it is still working (an actively-running no-mistakes step, -# or a backend busy signal), and surfaced otherwise, so a crew that finishes -# without a current working signal is never silently swallowed. A declared -# external-wait pause is the separate idle absorb case and re-surfaces only on -# its long bounded cadence. A repaint-only repeat of an already-surfaced keyed -# open-decision set is also absorbed, but a confidently dead parked agent still -# enters the wedge timer. The initial no-verb status signal still surfaces in -# normal mode. +# POSITIVE evidence it is still working (a current or bounded-degraded +# no-mistakes run step, or a backend busy signal), and surfaced otherwise, so a +# crew that finishes (or stops and waits) without a current working signal is +# never silently swallowed. A declared external-wait pause is the separate idle +# absorb case and re-surfaces only on its long bounded cadence. A repaint-only +# repeat of an already-surfaced keyed open-decision set is also absorbed, but a +# confidently dead parked agent still enters the wedge timer. The initial no-verb +# status signal still surfaces in normal mode. # While state/.afk exists, the daemon owns triage and this watcher queues and exits # on every wake. Printed reason lines: # signal: ... status/turn-end signals, surfaced when a listed status @@ -19,7 +19,8 @@ # is not provably working, unless afk is active # stale: a provably-working stale is ALWAYS absorbed (with a wedge # timer) regardless of what the status log says - an active -# run-step or busy pane outranks even a captain-relevant log +# current or bounded-degraded run step, or a busy pane, +# outranks even a captain-relevant log # line, since the crew's own log gets no new entry once # firstmate hands it to a no-mistakes validation. A declared # external-wait pause is absorbed instead with its own long @@ -132,7 +133,7 @@ SIGNAL_GRACE=${FM_SIGNAL_GRACE:-30} # seconds to linger after a signal so trai # debug log, and keeps blocking WITHOUT enqueuing or exiting. The no-verb signal # / first-sighting stale path is absorb-only-when-provably-working: such a wake # is absorbed ONLY while the crew shows positive evidence it is still working -# (an actively-running no-mistakes step, or a busy pane, via +# (a current or bounded-degraded no-mistakes run step, or a busy pane, via # crew_is_provably_working over fm-crew-state.sh); a crew that stopped its turn # with no running pipeline and no # busy pane is SURFACED, so a finish reported only through interactive pane menus @@ -1085,9 +1086,10 @@ EOF # poll. Root cause of the 2026-07 herdr false-surface incidents: a # validating crew was surfaced as stale every few minutes despite an # actively-running pipeline, purely because of this stale leftover - # line. On a NEW hash, give an active run/busy pane (the same - # authoritative source fm-crew-state.sh itself already prioritizes - # over the log) a chance to override before trusting the log. + # line. On a NEW hash, give a working current or bounded-degraded + # run/busy pane (the same authoritative source fm-crew-state.sh itself + # already prioritizes over the log) a chance to override before + # trusting the log. # # Key this one-shot on the complete open-decision set, not volatile # pane bytes. mark_surfaced reconciles this marker for every surfaced diff --git a/docs/architecture.md b/docs/architecture.md index 88a8ca37fee..5d9145d0458 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -18,7 +18,7 @@ When a canonical validated PR poll returns exactly `merged`, the watcher appends The receipt makes retirement safely retryable across restarts: fixed-path recovery revalidates the same evidence, removes the runnable check first, removes its registration and data sidecars, removes the receipt last, and preserves task metadata including `pr=` and `pr_head=`. A concurrent replacement remains armed, every non-merged or invalid observation remains unchanged, and retirement never performs task or persistent-secondmate cleanup. `bin/fm-pr-lib.sh` owns the receipt format and strict identity mechanics, while `bin/fm-watch.sh` owns queue-before-retirement ordering. -No-verb wakes, such as `working:` notes and bare turn-ended signals, are benign only when `bin/fm-crew-state.sh` reports positive evidence that the crew is still working: an actively running no-mistakes step attributed to that crew's current code, or an exact busy verdict from the semantic busy-state contract. +No-verb wakes, such as `working:` notes and bare turn-ended signals, are benign only when `bin/fm-crew-state.sh` reports positive evidence that the crew is still working: an attributed active no-mistakes step, its recent age-bounded degraded replay after a lookup failure, or an exact busy verdict from the semantic busy-state contract. A crew that declares `paused:` for a known external wait is separately absorbed while idle and re-surfaced only on the longer pause cadence, rather than being treated as a possible wedge. For an ordinary crew that has stopped, the normal-mode watcher first surfaces one stale wake, then applies that same cadence to an unchanged `paused:` or durable `captain-held` endpoint only when the backend confidently reports its agent dead. Live or inconclusive liveness remains fail-open at that initial surface, and the secondmate idle-endpoint exemption is unchanged. @@ -33,11 +33,14 @@ After each drain, `fm-wake-drain.sh` runs the same liveness guard as the supervi Routine watcher polling, supervision no-ops, elapsed waiting time, and absorbed benign wakes stay silent. A declared external wait trades that silence for one bounded recheck per pause window, so a forgotten pause cannot remain invisible indefinitely. Crew status files are append-only wake-event logs, not current-state fields. -`bin/fm-crew-state.sh ` is the cheap current-state read for an actionable heartbeat review: it attributes a no-mistakes run, active or terminal, only when it matches the crew's branch and current code identity, then keeps that run-step authoritative even if the pane has closed. -The script header owns the exact run-head ancestry rules. +`bin/fm-crew-state.sh ` is the cheap current-state read for an actionable heartbeat review: it attributes a no-mistakes run to the crew's branch while reconciling the run head with pipeline-owned commit behavior, then keeps that run-step authoritative even if the pane has closed. +The script header owns the exact run-head relationship, current-versus-historical source, lookup-failure, cache, and source-precedence rules. +Architecturally, an in-progress pipeline may advance the crew tip or report a head unavailable in that worktree without losing attribution, while the narrowed exceptions continue to reject stale historical evidence. +A run lookup that cannot complete remains distinct from a confirmed absence and may replay that crew's recent observed step under source `run-step-degraded` only within a finite window after endpoint-liveness and exact-busy checks, preserving wedge detection when the crew stops or the outage persists. During no-mistakes' `ci` monitor phase, it also reads the ci step log tail because `axi status` reports both "still waiting on checks" and "checks green, waiting on merge" as `ci,running`. The most recent recognized ci log marker wins, so checks-green monitoring reports done while a later re-arm, failed-check, or issue marker returns the crew to working. -Only when no matching run exists does it consult semantic busy state; exact busy reports working, exact idle permits fallback to a status-log event whose verb maps to a recognized run-state, and unknown or a dead pane stays unknown instead of trusting a stale log. +When no run is attributed, an exact busy verdict reports working before any degraded or fallback source. +After a failed lookup, a usable degraded record then outranks idle or unavailable harness evidence and the status log; without that record, or after a completed lookup, exact idle permits fallback to a status-log event whose verb maps to a recognized run-state, while unknown or a dead pane stays unknown instead of trusting a stale log. Decision-only events such as `resolved` never become current state or leak their prose into the current-state detail. In that status-log fallback, a declared external wait reports the distinct `paused` state with its reason. The semantic branch reports working only on an exact busy verdict and names the source that produced it; an unknown verdict never becomes working, never permits the status-log fallback, and never becomes a silent idle. diff --git a/docs/configuration.md b/docs/configuration.md index b3b3dbc59e3..cfcde2d04ca 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -516,7 +516,8 @@ FM_PROCEVENT_MAX_OUTPUT_BYTES=1048576 # bound on one captured process-to-event FM_PROCEVENT_CLAIM_ROOT= # machine-wide source claim root; default $XDG_STATE_HOME/firstmate/procevent-claims FM_CODEX_WATCH_CHECKPOINT=180 # seconds per foreground watcher checkpoint in Codex primary supervision FM_CREW_STATE_NM_TIMEOUT=10 # seconds allowed per no-mistakes query inside fm-crew-state.sh -FM_CREW_STATE_RUNS_LIMIT=200 # recent no-mistakes run rows scanned when axi status cannot be attributed to the current code +FM_CREW_STATE_RUNS_LIMIT=200 # recent no-mistakes run rows scanned when axi status cannot be attributed to the worktree +FM_CREW_STATE_DEGRADED_MAX_AGE=900 # seconds a recorded run-step may stand in as the degraded answer while the no-mistakes lookup cannot complete; 0 disables the degrade FM_CREW_STATE_BIN=bin/fm-crew-state.sh # test override for the current-state reader used by working/paused watcher triage FMX_PAIRING_TOKEN= # X mode pairing token; .env opt-in authorizes replies and eligible lifecycle actions FMX_RELAY_URL=https://myfirstmate.io # optional X relay override, mainly for local relay development diff --git a/tests/fm-crew-state.test.sh b/tests/fm-crew-state.test.sh index 8f986b6139e..dbaf80890ad 100755 --- a/tests/fm-crew-state.test.sh +++ b/tests/fm-crew-state.test.sh @@ -4,9 +4,10 @@ # # The status file (state/.status) is a best-effort append-only EVENT LOG, so # `tail -1` of it reports the last event, not the current state. fm-crew-state -# reads the AUTHORITATIVE source (a matching no-mistakes run-step, else the -# semantic busy-state contract) and reconciles the possibly-stale log against it. These -# cases pin every branch of that logic, hermetically, over real throwaway git +# reconciles an attributable no-mistakes run-step, a recent recorded step when +# that lookup fails, and the semantic busy-state contract before consulting the +# possibly-stale log. These cases pin every branch of that logic, hermetically, +# over real throwaway git # repos with a fake `no-mistakes` (run-step source) and a fake `tmux` (pane # source): # (a) active run-step is authoritative -> run-step @@ -25,6 +26,9 @@ # This is the direct regression pair for the 2026-07-02 herdr incident, # proving the watcher's own absorb-only-when-provably-working predicate # benefits from the fix in both directions. +# (l) failed lookup reuses only a recent observed run step -> run-step-degraded +# (m) an executing run keeps its pipeline-authored tip advance -> run-step +# (n) branch-scoped live status accepts a pipeline-owned head -> run-step set -u # shellcheck source=tests/lib.sh @@ -62,6 +66,12 @@ make_fakebin() { # -> echoes fakebin path cat > "$fb/no-mistakes" <<'SH' #!/usr/bin/env bash set -u +# A lookup that cannot COMPLETE, as opposed to one that completes and finds no +# run. FM_FAKE_NM_SLEEP outlasts the helper's own bound so the REAL timeout +# wrapper kills the call; FM_FAKE_NM_RC returns a timeout's exit status directly +# for the cases that only need the failure, not the wall-clock wait. +[ -n "${FM_FAKE_NM_SLEEP:-}" ] && sleep "$FM_FAKE_NM_SLEEP" +[ "${FM_FAKE_NM_RC:-0}" = 0 ] || exit "${FM_FAKE_NM_RC}" case "${1:-}" in axi) shift @@ -75,6 +85,7 @@ case "${1:-}" in esac ;; runs) + [ "${FM_FAKE_RUNS_RC:-0}" = 0 ] || exit "${FM_FAKE_RUNS_RC}" printf '%s\n' "${FM_FAKE_RUNS_LIST:-}" ;; esac exit 0 @@ -149,6 +160,13 @@ new_case() { # -> echoes case dir with an empty state/ printf '%s\n' "$d" } +arm_busy_record() { # + local state=$1 id=$2 gen + gen=$("$ROOT/bin/fm-busy-event.sh" arm "$state" "$id") + "$ROOT/bin/fm-busy-event.sh" apply "$state" "$id" busy --gen "$gen" \ + --source claude-hook --event user-prompt-submit +} + arm_idle_record() { # local state=$1 id=$2 gen gen=$("$ROOT/bin/fm-busy-event.sh" arm "$state" "$id") @@ -170,8 +188,14 @@ reset_fakes() { FM_FAKE_HERDR_MISSING=0 FM_FAKE_HERDR_AGENT_STATUS="" FM_FAKE_CI_LOGS="" + FM_FAKE_NM_RC=0 + FM_FAKE_RUNS_RC=0 + FM_FAKE_NM_SLEEP="" + FM_CREW_STATE_DEGRADED_MAX_AGE="" + FM_CREW_STATE_NM_TIMEOUT="" export FM_FAKE_AXI_STATUS FM_FAKE_AXI_STATUS_RUN FM_FAKE_RUNS_LIST FM_FAKE_BUSY FM_FAKE_BUSY_TEXT FM_FAKE_TMUX_MISSING export FM_FAKE_HERDR_BUSY FM_FAKE_HERDR_MISSING FM_FAKE_HERDR_AGENT_STATUS FM_FAKE_CI_LOGS + export FM_FAKE_NM_RC FM_FAKE_RUNS_RC FM_FAKE_NM_SLEEP FM_CREW_STATE_DEGRADED_MAX_AGE FM_CREW_STATE_NM_TIMEOUT } # --- run-object fixtures (TOON, as `no-mistakes axi status` emits) ----------- @@ -265,6 +289,42 @@ steps[3]{step,status,findings,duration_ms}: EOF } +# The pipeline owns the branch and is committing in its own copy, so the head it +# reports is not an object this crew's worktree has ever seen. This fixture keeps +# the live `axi status` shape for a run parked at fix_review. +run_pipeline_owned_fix_review() { # + cat < + cat < cat </dev/null fm_write_meta "$d/state/feat-ci.meta" "window=fm:fm-feat-ci" "worktree=$d/wt" "kind=ship" + seed_known_run_step "$d" feat-ci fm/feat-ci printf 'done: PR https://github.com/o/r/pull/2 checks green\n' > "$d/state/feat-ci.status" FM_FAKE_AXI_STATUS="$(run_ci_monitoring fm/feat-ci)" + arm_idle_record "$d/state" feat-ci local out; out=$(run_crew_state "$d" feat-ci) assert_contains "$out" "state: done" "ci-ready status log -> done" assert_contains "$out" "source: status-log" "ci-ready state comes from the status log" assert_contains "$out" "checks green" "ci-ready detail preserves the report" assert_not_contains "$out" "state: working" "ci-ready is not hidden by monitoring run" - pass "ci-ready status log beats monitoring run" + FM_FAKE_NM_RC=124 + out=$(run_crew_state "$d" feat-ci) + assert_contains "$out" "state: done" "a later timeout preserves the CI-ready verdict" + assert_contains "$out" "source: run-step-degraded" "a later timeout degrades from the CI-ready verdict" + PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" crew_is_provably_working feat-ci \ + && fail "a timed-out lookup replayed stale working evidence after CI became ready" + pass "ci-ready status log updates the record before a later timeout" } # Regression for the PR #252 incident: the crew's own status log never got a @@ -1290,7 +1358,7 @@ test_local_advanced_past_run_head_invalidates() { pass "local work advanced past run head invalidates attribution" } -test_missing_run_head_falls_back_to_current_state() { +test_missing_run_head_with_failed_fallback_degrades() { reset_fakes local d out d=$(new_case missing-run-head) @@ -1298,15 +1366,372 @@ test_missing_run_head_falls_back_to_current_state() { make_fakebin "$d" >/dev/null fm_write_meta "$d/state/no-head.meta" "window=fm:fm-no-head" "worktree=$d/wt" "kind=ship" "harness=claude" printf 'working: current stage still in progress\n' > "$d/state/no-head.status" + seed_known_run_step "$d" no-head fm/feat-no-head FM_FAKE_AXI_STATUS=$(run_parked fm/feat-no-head | grep -v '^ head:') - FM_FAKE_RUNS_LIST="" + FM_FAKE_RUNS_RC=124 FM_FAKE_BUSY=0 arm_idle_record "$d/state" no-head out=$(run_crew_state "$d" no-head) - assert_not_contains "$out" "source: run-step" "missing run head must not permit branch-only attribution" - assert_contains "$out" "source: status-log" "missing run head falls back to current state sources" - assert_contains "$out" "state: working" "status-log remains current after missing run head" - pass "missing run head falls back instead of matching by branch" + assert_contains "$out" "source: run-step-degraded" "missing run head preserves the recorded step" + assert_contains "$out" "state: working" "missing run head stays known after fallback failure" + PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" crew_is_provably_working no-head \ + || fail "an inconclusive missing head erased the crew's known working step" + pass "missing run head degrades when the fallback lookup fails" +} + +# --------------------------------------------------------------------------- +# (l) A run lookup that could not COMPLETE must not read like a run ABSENCE. +# These cases reproduce a validating crew whose bounded no-mistakes lookup times +# out, then guard the evidence and age bounds that keep a real wedge visible. + +# A crew whose state is read once while validating, then read again while the +# lookup times out, stays at its last known step instead of going unknown. +seed_known_run_step() { # + local out + FM_FAKE_AXI_STATUS="$(run_running "$3")" + out=$(run_crew_state "$1" "$2") + assert_contains "$out" "source: run-step" "seed read must establish a real run-step" +} + +test_lookup_timeout_degrades_to_last_known_run_step() { + reset_fakes + local d out; d=$(new_case lookup-timeout-degrades) + make_repo_on_branch "$d/wt" fm/feat-degraded + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/degraded.meta" "window=fm:fm-degraded" "worktree=$d/wt" "kind=ship" "harness=claude" + seed_known_run_step "$d" degraded fm/feat-degraded + # Now the daemon stops answering within the bound, and the pane looks idle - + # exactly the shape that used to collapse to unknown/none. + FM_FAKE_NM_RC=124 + FM_FAKE_BUSY=0 + arm_idle_record "$d/state" degraded + out=$(run_crew_state "$d" degraded) + assert_not_contains "$out" "state: unknown" "a timed-out lookup must not read as unknown" + assert_contains "$out" "state: working" "degrades to the last known run-step state" + assert_contains "$out" "source: run-step-degraded" "degraded answer is labelled as such" + assert_contains "$out" "run lookup unavailable" "degraded answer names why it is degraded" + assert_contains "$out" "validating (running)" "degraded answer replays the last known step" + pass "a timed-out run lookup degrades to the last known run-step, not unknown" +} + +# The same property through the REAL timeout wrapper rather than a canned exit +# code, so the bound itself is proven to produce the degraded path. +test_real_timeout_bound_degrades_to_last_known_run_step() { + reset_fakes + local d out; d=$(new_case lookup-timeout-real) + make_repo_on_branch "$d/wt" fm/feat-realtimeout + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/realtimeout.meta" "window=fm:fm-realtimeout" "worktree=$d/wt" "kind=ship" "harness=claude" + seed_known_run_step "$d" realtimeout fm/feat-realtimeout + FM_CREW_STATE_NM_TIMEOUT=1 + FM_FAKE_NM_SLEEP=5 + FM_FAKE_BUSY=0 + arm_idle_record "$d/state" realtimeout + out=$(run_crew_state "$d" realtimeout) + assert_contains "$out" "source: run-step-degraded" "a genuinely killed query degrades" + assert_contains "$out" "state: working" "genuinely killed query keeps the last known state" + pass "the real timeout bound produces the degraded run-step, not unknown" +} + +# SAFETY: the degrade must never invent evidence. A crew that stopped before any +# run was ever observed has nothing on record and must still surface. +test_lookup_timeout_without_known_run_step_still_surfaces() { + reset_fakes + local d out; d=$(new_case lookup-timeout-no-record) + make_repo_on_branch "$d/wt" fm/feat-norecord + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/norecord.meta" "window=fm:fm-norecord" "worktree=$d/wt" "kind=ship" "harness=claude" + FM_FAKE_NM_RC=124 + FM_FAKE_BUSY=0 + arm_idle_record "$d/state" norecord + out=$(run_crew_state "$d" norecord) + assert_not_contains "$out" "run-step-degraded" "no recorded step means nothing to degrade to" + assert_contains "$out" "state: unknown" "a stopped crew with no observed run still surfaces" + PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" crew_is_provably_working norecord \ + && fail "a crew that never validated was treated as provably working on a failed lookup" + pass "a failed lookup with no recorded run-step still surfaces the crew" +} + +# SAFETY: the degrade is time-bounded, so a permanently unreachable daemon stops +# standing in for evidence instead of hiding a real wedge forever. +test_degraded_run_step_expires() { + reset_fakes + local d out; d=$(new_case degraded-expires) + make_repo_on_branch "$d/wt" fm/feat-expires + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/expires.meta" "window=fm:fm-expires" "worktree=$d/wt" "kind=ship" "harness=claude" + seed_known_run_step "$d" expires fm/feat-expires + FM_FAKE_NM_RC=124 + FM_FAKE_BUSY=0 + FM_CREW_STATE_DEGRADED_MAX_AGE=0 + arm_idle_record "$d/state" expires + out=$(run_crew_state "$d" expires) + assert_not_contains "$out" "run-step-degraded" "an expired record must not be replayed" + assert_contains "$out" "state: unknown" "an expired record surfaces the crew again" + PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" crew_is_provably_working expires \ + && fail "an expired degraded record was still treated as provably working" + pass "the degraded run-step expires instead of absorbing wedge suspicion forever" +} + +# SAFETY: a dead endpoint is positive evidence the crew stopped, and must keep +# outranking a recent memory of it validating. +test_degraded_run_step_not_used_when_endpoint_gone() { + reset_fakes + local d out; d=$(new_case degraded-endpoint-gone) + make_repo_on_branch "$d/wt" fm/feat-gone + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/gone.meta" "window=fm:fm-gone" "worktree=$d/wt" "kind=ship" "harness=claude" + seed_known_run_step "$d" gone fm/feat-gone + FM_FAKE_NM_RC=124 + FM_FAKE_TMUX_MISSING=1 + out=$(run_crew_state "$d" gone) + assert_not_contains "$out" "run-step-degraded" "a gone endpoint must not replay a recorded step" + assert_contains "$out" "state: unknown" "a gone endpoint still reports unknown" + assert_contains "$out" "backend target gone" "the gone endpoint is the reported reason" + PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" crew_is_provably_working gone \ + && fail "a crew whose endpoint is gone was treated as provably working" + pass "a gone endpoint outranks the recorded run-step" +} + +# SAFETY: absence is not failure. A lookup that COMPLETES and finds no run for +# this branch must fall through to the live sources, never replay the record. +test_completed_lookup_without_run_does_not_degrade() { + reset_fakes + local d out; d=$(new_case absence-not-failure) + make_repo_on_branch "$d/wt" fm/feat-absent + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/absent.meta" "window=fm:fm-absent" "worktree=$d/wt" "kind=ship" "harness=claude" + seed_known_run_step "$d" absent fm/feat-absent + # The CLI answers normally; the run for this branch is simply gone from both + # the bare status answer and the runs list. + FM_FAKE_AXI_STATUS="$(run_running fm/other-crew)" + FM_FAKE_RUNS_LIST="$(cat <<'EOF' + running fm/other-crew aaaaaaa 2026-08-01 22:10 +EOF +)" + FM_FAKE_BUSY=0 + arm_idle_record "$d/state" absent + out=$(run_crew_state "$d" absent) + assert_not_contains "$out" "run-step-degraded" "a completed lookup must not degrade" + assert_contains "$out" "state: unknown" "a completed lookup with no run falls through" + PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" crew_is_provably_working absent \ + && fail "a completed lookup that found no run was treated as provably working" + FM_FAKE_NM_RC=124 + out=$(run_crew_state "$d" absent) + assert_not_contains "$out" "run-step-degraded" "a later timeout must not resurrect the absent run" + assert_contains "$out" "state: unknown" "a later timeout still surfaces the crew after confirmed absence" + PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" crew_is_provably_working absent \ + && fail "a timed-out lookup resurrected a run invalidated by confirmed absence" + pass "a completed lookup invalidates the recorded step before a later timeout" +} + +# SAFETY: live positive evidence outranks a remembered step. A harness that is +# busy right now is better current-state evidence than any record. +test_busy_pane_outranks_degraded_record() { + reset_fakes + local d out; d=$(new_case degraded-vs-busy) + make_repo_on_branch "$d/wt" fm/feat-degvbusy + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/degvbusy.meta" "window=fm:fm-degvbusy" "worktree=$d/wt" "kind=ship" "harness=claude" + # Last observed step was terminal, so a replayed record would say done. + FM_FAKE_AXI_STATUS="$(run_passed fm/feat-degvbusy)" + out=$(run_crew_state "$d" degvbusy) + assert_contains "$out" "state: done" "seed read must record the terminal step" + FM_FAKE_NM_RC=124 + FM_FAKE_BUSY=1 + arm_busy_record "$d/state" degvbusy + out=$(run_crew_state "$d" degvbusy) + assert_contains "$out" "source: pane" "a busy harness outranks the recorded step" + assert_contains "$out" "state: working" "a busy harness reads working, not the remembered done" + pass "a live busy harness outranks the recorded run-step" +} + +# The degraded step still outranks the append-only status log, which is the +# reason this reader exists: a resolved gate leaves its needs-decision line +# behind forever, and a remembered run-step must not lose to it. +test_degraded_record_outranks_stale_status_log() { + reset_fakes + local d out; d=$(new_case degraded-vs-log) + make_repo_on_branch "$d/wt" fm/feat-degvlog + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/degvlog.meta" "window=fm:fm-degvlog" "worktree=$d/wt" "kind=ship" "harness=claude" + printf 'needs-decision: review gate\n' > "$d/state/degvlog.status" + seed_known_run_step "$d" degvlog fm/feat-degvlog + FM_FAKE_NM_RC=124 + FM_FAKE_BUSY=0 + arm_idle_record "$d/state" degvlog + out=$(run_crew_state "$d" degvlog) + assert_contains "$out" "source: run-step-degraded" "the recorded step outranks the status log" + assert_contains "$out" "state: working" "a resolved gate's stale line must not resurface as parked" + pass "the degraded run-step outranks a stale status-log line" +} + +# --------------------------------------------------------------------------- +# (n) Pipeline-owned head: a branch-scoped live run may report a head this +# worktree cannot resolve because the pipeline commits in its own copy. This is +# distinct from both lookup failure and a resolvable head that diverged. + +# A non-resolvable sha of the right shape. Deliberately not an object anywhere. +UNRESOLVABLE_HEAD=1c17405cdeadbeefdeadbeefdeadbeefdeadbeef + +test_pipeline_owned_unresolvable_head_attributes() { + reset_fakes + local d out; d=$(new_case pipeline-owned-head) + make_repo_on_branch "$d/wt" fm/feat-pipeowned + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/pipeowned.meta" "window=fm:fm-pipeowned" "worktree=$d/wt" \ + "kind=ship" "harness=claude" + git -C "$d/wt" rev-parse --verify -q "$UNRESOLVABLE_HEAD^{commit}" >/dev/null 2>&1 \ + && fail "fixture head must not resolve in the worktree" + FM_FAKE_RUN_HEAD="$UNRESOLVABLE_HEAD" + FM_FAKE_AXI_STATUS="$(run_pipeline_owned_fix_review fm/feat-pipeowned)" + FM_FAKE_BUSY=0 + arm_idle_record "$d/state" pipeowned + out=$(run_crew_state "$d" pipeowned) + assert_contains "$out" "source: run-step" "a live run whose head this worktree cannot see still attributes" + assert_contains "$out" "state: parked" "the pipeline-owned run reports its real gate" + assert_contains "$out" "parked at review" "the gate is named" + assert_contains "$out" "3 finding(s)" "the gate finding count survives" + pass "a live run whose head the worktree cannot resolve still attributes to its branch" +} + +# SAFETY: only a LIVE run may claim an unresolvable head. A finished run whose +# head this worktree never saw is not evidence of anything current. +test_terminal_unresolvable_head_still_rejected() { + reset_fakes + local d out; d=$(new_case terminal-unresolvable-head) + make_repo_on_branch "$d/wt" fm/feat-termunres + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/termunres.meta" "window=fm:fm-termunres" "worktree=$d/wt" \ + "kind=ship" "harness=claude" + printf 'working: stage 2 implementation in progress\n' > "$d/state/termunres.status" + seed_known_run_step "$d" termunres fm/feat-termunres + FM_FAKE_RUN_HEAD="$UNRESOLVABLE_HEAD" + FM_FAKE_AXI_STATUS="$(run_terminal_unresolvable_head fm/feat-termunres)" + FM_FAKE_RUNS_RC=124 + FM_FAKE_BUSY=0 + arm_idle_record "$d/state" termunres + out=$(run_crew_state "$d" termunres) + assert_not_contains "$out" "source: run-step" "a finished run with an unseen head must not attribute" + assert_contains "$out" "source: status-log" "falls back to the current-state sources" + PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" crew_is_provably_working termunres \ + && fail "a terminal same-branch answer was hidden by a failed historical lookup" + pass "a terminal same-branch answer invalidates the record before fallback failure" +} + +# SAFETY: the runs LISTING is a historical log with no notion of "current", so an +# unresolvable sha there stays too weak to attribute even on a running row. Only +# the branch-scoped `axi status` answer earns that trust. +test_runs_list_unresolvable_head_still_rejected() { + reset_fakes + local d out; d=$(new_case coarse-unresolvable-head) + make_repo_on_branch "$d/wt" fm/feat-coarseunres + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/coarseunres.meta" "window=fm:fm-coarseunres" "worktree=$d/wt" \ + "kind=ship" "harness=claude" + printf 'working: stage 2 implementation in progress\n' > "$d/state/coarseunres.status" + FM_FAKE_AXI_STATUS="$(run_running fm/other-crew)" + FM_FAKE_RUNS_LIST="$(cat </dev/null + fm_write_meta "$d/state/provdeg.meta" "window=fm:fm-provdeg" "worktree=$d/wt" "kind=ship" "harness=claude" + seed_known_run_step "$d" provdeg fm/feat-provable-degraded + FM_FAKE_NM_RC=124 + FM_FAKE_BUSY=0 + arm_idle_record "$d/state" provdeg + PATH="$d/fakebin:$PATH" FM_STATE_OVERRIDE="$d/state" crew_is_provably_working provdeg \ + || fail "a validating crew whose lookup timed out was not treated as provably working" + pass "crew_is_provably_working absorbs a validating crew whose run lookup timed out" +} + +# --------------------------------------------------------------------------- +# (m) Fix-round attribution: a run that advances the tip with its OWN commits. +# The code-identity skip is right for a stale or rewritten run and wrong for the +# actively executing run currently authoring those commits. + +test_coarse_running_row_behind_tip_does_not_attribute() { + reset_fakes + local d base_short out; d=$(new_case coarse-running-behind) + make_repo_on_branch "$d/wt" fm/feat-coarse-behind + base_short=$(git -C "$d/wt" rev-parse --short=8 HEAD) + git -C "$d/wt" commit -q --allow-empty -m 'local work after parked run' + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/coarsebehind.meta" "window=fm:fm-coarsebehind" "worktree=$d/wt" "kind=ship" "harness=claude" + FM_FAKE_AXI_STATUS="$(run_running fm/other-crew)" + FM_FAKE_RUNS_LIST="$(cat </dev/null + fm_write_meta "$d/state/fixaxi.meta" "window=fm:fm-fixaxi" "worktree=$d/wt" "kind=ship" "harness=claude" + # The run still reports the head it recorded before committing the fix. + FM_FAKE_RUN_HEAD="$base_head" + FM_FAKE_AXI_STATUS="$(run_fixing fm/feat-fixround-axi)" + FM_FAKE_BUSY=0 + arm_idle_record "$d/state" fixaxi + out=$(run_crew_state "$d" fixaxi) + assert_contains "$out" "source: run-step" "an active run behind its own fix commit keeps attribution" + assert_contains "$out" "validating (fixing)" "the fix round reports its own step" + pass "an active axi-status run behind its own fix commits still attributes" +} + +# SAFETY: only an ACTIVELY-EXECUTING run may claim an advanced tip. A terminal +# run whose branch moved on afterwards is exactly the stale run the skip exists +# for, and must still be rejected. +test_terminal_run_behind_advanced_tip_still_invalidates() { + reset_fakes + local d base_short out; d=$(new_case terminal-behind-tip) + make_repo_on_branch "$d/wt" fm/feat-terminal-behind + base_short=$(git -C "$d/wt" rev-parse --short=8 HEAD) + git -C "$d/wt" commit -q --allow-empty -m 'local stage-2 work after the run finished' + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/terminalbehind.meta" "window=fm:fm-terminalbehind" \ + "worktree=$d/wt" "kind=ship" "harness=claude" + printf 'working: stage 2 implementation in progress\n' > "$d/state/terminalbehind.status" + FM_FAKE_AXI_STATUS="$(run_running fm/other-crew)" + FM_FAKE_RUNS_LIST="$(cat <