diff --git a/.agents/skills/operational-home-layout/SKILL.md b/.agents/skills/operational-home-layout/SKILL.md index f51340058bb..1ca3277f56b 100644 --- a/.agents/skills/operational-home-layout/SKILL.md +++ b/.agents/skills/operational-home-layout/SKILL.md @@ -67,6 +67,7 @@ state/ runtime records and signals; gitignored .devin-config.json firstmate-owned per-task Devin config (mode 600 snapshot of the user config plus the busy-state and turn-end hooks) passed through --config so no user or project config is edited; bin/fm-devin-config.sh owns it; removed by teardown .muse-session muse busy-source binding (sessions root plus task worktree) written by fm-spawn; removed by teardown .cursor-session cursor busy-source binding (projects root, task worktree, prior conversations) written by fm-spawn; removed by teardown + .worker-state exact record of a deliberately stood-down ship or scout worker; bin/fm-worker-state-lib.sh owns its format and the watcher admits no suppression unless the endpoint remains provably worker-free .git-hooks/ per-task git hooksPath that strips AI commit trailers at the commit object unless config/keep-ai-trailers is present; written by fm-spawn, removed by teardown (bin/fm-git-strip-ai-trailers.sh) .reconcile-nudged epoch second of the last inventory-reconcile nudge sent to this secondmate; bin/fm-secondmate-reconcile.sh owns its per-home cooldown window .backlog-close the exact backlog transition a teardown recorded before removing the task's record, so an interrupted cleanup can still be finished at the next session start; bin/fm-backlog-transition-lib.sh owns its format and replay, and a landed transition removes it diff --git a/AGENTS.md b/AGENTS.md index 507a7f51498..216cfd5288b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -210,7 +210,7 @@ Steer a worker with ordinary text through fail-closed `fm-send`: the message bec A remote secondmate steer rides the same durable-inbox model through the remote transport; after an unconfirmed delivery, only the exact `FM_PENDING_REPLY_EXISTING_CORR=` resend command printed by `fm-send` is safe because it preserves the request body for remote enqueue deduplication (`bin/fm-send.sh` header). When a steer answers an open keyed decision or blocker, pass `fm-send`'s `--resolve-key` so the answer itself closes that decision record at answer time, identically for local and remote workers (contract: `bin/fm-send.sh` header). `fm-send` is the data plane for text the worker should read; never use its key or text paths for interrupt, exit, or other lifecycle control, because routing-marked lifecycle text becomes chat the worker reasons about instead of executing. -Drive a worker's lifecycle through `bin/fm-control.sh interrupt|exit|relaunch`, which owns the per-runtime mechanics, verifies each action, and never tears down or discards anything ([`docs/agent-control.md`](docs/agent-control.md)). +Drive a worker's lifecycle through `bin/fm-control.sh interrupt|exit|stand-down|repair-worker-state|relaunch`, which owns the per-runtime mechanics, verifies each action, and never tears down or discards anything ([`docs/agent-control.md`](docs/agent-control.md)). A secondmate's routed reply returns through status or a document pointer, not by firstmate peeking into its chat. For the parent-owned correlation, recovery, and escalation contract on marked secondmate requests, see `bin/fm-pending-reply-lib.sh`. When the captain adds or changes an ask mid-task, append the captain's words without added speaker labels or direct address to that brief's `## Captain's intent` and relay those words to the worker; Firstmate build constraints stay in `## Firstmate spec` or the steer. diff --git a/bin/fm-control-lib.sh b/bin/fm-control-lib.sh index d3fcbcb043d..f735ff6bd85 100644 --- a/bin/fm-control-lib.sh +++ b/bin/fm-control-lib.sh @@ -49,13 +49,15 @@ fm_control_verbs() { cat <<'EOF' interrupt exit +stand-down +repair-worker-state relaunch EOF } fm_control_verb_allowed() { # case "${1-}" in - interrupt|exit|relaunch) return 0 ;; + interrupt|exit|stand-down|repair-worker-state|relaunch) return 0 ;; esac return 1 } diff --git a/bin/fm-control.sh b/bin/fm-control.sh index aa4c9af2004..56ffd4dfe1c 100755 --- a/bin/fm-control.sh +++ b/bin/fm-control.sh @@ -4,6 +4,8 @@ # # Usage: fm-control.sh interrupt # fm-control.sh exit +# fm-control.sh stand-down +# fm-control.sh repair-worker-state # fm-control.sh relaunch [--harness ] [--model ] # [--effort ] # (--note | --note-file ) @@ -51,6 +53,34 @@ # endpoint, so this verb cannot tell a destroyed window from one on # a tmux server it cannot address, and it will not claim a stop it # cannot see. +# stand-down Stop the agent for a deliberately held ship or scout task, then +# atomically record that the task has no worker on purpose. The +# record is published only after the recovery-grade classifier +# proves the agent is gone, so a live worker never loses stale or +# wedge detection. The record is cleared when `relaunch` starts a +# replacement in the preserved +# endpoint and worktree, and restored if that relaunch aborts +# before the replacement is published. An interrupted transition +# remains `standing-down`, which stays under ordinary supervision. +# Refused while this task owns an in-flight no-mistakes run: that +# run owns the branch and needs a worker at its gates, and +# stand-down never cancels one for you. Refused too when that +# question cannot be answered at all, naming the check that could +# not answer, because the record it would publish suppresses +# supervision. Refused as well while an unacknowledged steering +# instruction is still waiting in the task's inbox, naming it: the +# held task would have no worker to read it, so the worker handles +# it or the operator withdraws it first. An agent that already +# exited can be declared intentional only when the task's status +# log already declares the hold (`paused:`/`captain-held:`), since +# deadness alone is the ambiguity this record exists to resolve. +# repair-worker-state +# Reconcile a worker-state record against the endpoint's observed +# reality, idempotently and without hand-editing. Reality wins, +# and only toward supervision: an unprovable record, or a +# declaration a live agent contradicts, is cleared and the +# discrepancy is reported. A dead endpoint alone never declares +# intent, so repair can never create a suppression. # relaunch Transactionally replace the running agent with a new one, in the # SAME worktree - and the same endpoint whenever that endpoint # still exists - on the same or a newly chosen @@ -113,9 +143,9 @@ # - A backend that cannot deliver the harness's interrupt key is refused # (Orca's terminal API has no Escape). # - `exit` and `relaunch` require a backend with a recovery-grade agent-state -# classifier (tmux, herdr), because without one the "the agent stopped" -# postcondition cannot be proven. zellij, orca, and cmux are refused rather -# than reported as successful blind. +# classifier (tmux or herdr). The worker-state verbs stand-down and +# repair-worker-state are supported only on tmux. zellij, orca, and cmux +# are refused rather than reported as successful blind. # - An ambiguous or unreadable endpoint state refuses; only a positively # classified state acts. # - A composer that visibly holds pending text refuses before an exit command @@ -168,6 +198,14 @@ DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" . "$SCRIPT_DIR/fm-backend.sh" # shellcheck source=bin/fm-busy-lib.sh . "$SCRIPT_DIR/fm-busy-lib.sh" +# shellcheck source=bin/fm-worker-state-lib.sh +. "$SCRIPT_DIR/fm-worker-state-lib.sh" +# shellcheck source=bin/fm-classify-lib.sh +. "$SCRIPT_DIR/fm-classify-lib.sh" +# shellcheck source=bin/fm-nm-run-lib.sh +. "$SCRIPT_DIR/fm-nm-run-lib.sh" +# shellcheck source=bin/fm-task-inbox-lib.sh +. "$SCRIPT_DIR/fm-task-inbox-lib.sh" # shellcheck source=bin/fm-control-lib.sh . "$SCRIPT_DIR/fm-control-lib.sh" # shellcheck source=bin/fm-pr-lib.sh @@ -183,6 +221,10 @@ ARM_WAIT=${FM_CONTROL_ARM_WAIT:-1.5} EXIT_WAIT=${FM_CONTROL_EXIT_WAIT:-30} LAUNCH_WAIT=${FM_CONTROL_LAUNCH_WAIT:-90} EXIT_RETRIES=${FM_CONTROL_EXIT_RETRIES:-3} +# Bounded budget for the one read-only `axi status` question stand-down asks +# before it may declare a hold (bin/fm-nm-run-lib.sh owns the attribution). +NM_CONTROL_TIMEOUT=${FM_CONTROL_NM_TIMEOUT:-10} +case "$NM_CONTROL_TIMEOUT" in ''|*[!0-9]*) NM_CONTROL_TIMEOUT=10 ;; esac die() { # echo "error: $1" >&2 @@ -390,6 +432,11 @@ require_state_verified_backend() { # die "task $ID runs on the $BACKEND backend, which has no recovery-grade agent-state classifier, so '$1' cannot prove the agent actually stopped; refusing rather than reporting an unproven transition as done" } +require_worker_lifecycle_backend() { # + [ "$BACKEND" = tmux ] && return 0 + die "task $ID runs on the $BACKEND backend, where '$1' worker-lifecycle control is not supported; use the backend's authorised lifecycle path instead" +} + # rendered_matches : whether any row of the visible viewport matches. # An unreadable viewport is a no, so every caller treats it as missing proof. rendered_matches() { # @@ -654,6 +701,179 @@ do_exit() { printf 'stopped' } +# do_stand_down: turn a proven stopped agent into an explicit, intentional +# no-worker state. The transitional record is deliberately not a watcher +# exemption. That makes a crash or interruption between the requested stop and +# its proof visible instead of silently declaring a live or ambiguous task +# healthy. +# +# Three preconditions keep the declaration honest rather than inferred: +# - No in-flight no-mistakes run may be attributed to this task. An active +# run owns the branch and needs a worker to answer its gates, so a hold is +# not the operator's to declare yet. Stand-down never cancels a run: the +# operator finishes it or aborts it explicitly first. +# - An agent that is ALREADY gone can be declared intentional only when the +# task's own authoritative hold declaration says so (the status log's +# paused/captain-held verb, the same declaration both supervisors already +# honour, per bin/fm-classify-lib.sh). Deadness alone is exactly the +# ambiguity this record exists to resolve, so it never proves intent, and +# the hold is reversible by the ordinary next status append. +# - No unacknowledged steering instruction may be waiting for this worker. +# The held task would have no worker to read it, so a message nobody has +# read yet must first be handled or explicitly withdrawn. +do_stand_down() { + local lifecycle state result + [ "$KIND" != secondmate ] \ + || die "task $ID is a secondmate home; stand-down is only for held ship or scout work because a stopped secondmate would leave its own fleet unsupervised" + require_worker_lifecycle_backend stand-down + require_state_verified_backend stand-down + refuse_stand_down_during_active_run + refuse_stand_down_with_pending_instruction + lifecycle=$(fm_worker_state_status "$STATE" "$ID" "$T") + case "$lifecycle" in + invalid) + die "task $ID has an invalid worker-state record; run 'fm-control $ID repair-worker-state' to reconcile it against the endpoint before declaring a hold" + ;; + stood-down) + state=$(agent_state) + case "$state" in + dead) printf 'already-stood-down'; return 0 ;; + alive) die "task $ID is recorded as stood down but its agent is alive; refusing to hide a live worker from wedge detection (repair it with 'fm-control $ID repair-worker-state')" ;; + *) die "task $ID is recorded as stood down but its endpoint reads '$state'; inspect the endpoint before changing its worker state" ;; + esac + ;; + active) + state=$(agent_state) + case "$state" in + alive) + fm_worker_state_write "$STATE" "$ID" "$T" standing-down \ + || die "could not record the stand-down transition for task $ID; its worker remains under ordinary supervision" + ;; + dead) + task_is_declared_held \ + || die "task $ID's agent is already dead without a stand-down record; refusing to relabel a possible worker failure as intentional. Declare the hold first by appending a '$(fm_control_hold_verb): ' line to $STATE/$ID.status, then re-run stand-down" + ;; + missing|ambiguous|unreadable|unverified|*) + die "task $ID's endpoint reads '$state'; refusing to declare a worker-free hold without proof that the requested stop owns this state" + ;; + esac + ;; + standing-down) + state=$(agent_state) + case "$state" in + dead) ;; + alive) do_exit >/dev/null ;; + *) die "task $ID's interrupted stand-down reads '$state'; it remains under ordinary supervision until the endpoint can be reconciled" ;; + esac + ;; + esac + if [ "$lifecycle" = active ] && [ "$state" = alive ]; then + result=$(do_exit) + case "$result" in + stopped|already-stopped) ;; + *) die "task $ID's stand-down exit reported '$result'; refusing to publish an intentional no-worker state" ;; + esac + fi + fm_worker_state_write "$STATE" "$ID" "$T" stood-down \ + || die "task $ID's agent is stopped but its intentional worker-state record could not be published; it remains under ordinary supervision" + printf 'stood-down' +} + +# The declared-hold verb an operator uses to make an ordinary prior exit +# intentional. Named from the classifier's own configurable vocabulary so the +# refusal message can never drift from the verb the check accepts. +fm_control_hold_verb() { + printf '%s' "${FM_CLASSIFY_PAUSED_VERB:-$FM_CLASSIFY_PAUSED_VERB_DEFAULT}" +} + +# 0 when the task's authoritative current declaration is a deliberate hold. +task_is_declared_held() { + status_is_paused_or_captain_held "$(last_status_line "$STATE/$ID.status")" +} + +# Refuse a stand-down while this task owns an in-flight no-mistakes run. The +# run owns the branch and expects a worker at its gates, and stand-down must +# never cancel one on the operator's behalf. +# +# THE GOVERNING RULE HERE: IF IT CANNOT BE PROVEN SAFE, IT IS REFUSED, AND THE +# REFUSAL NAMES WHAT COULD NOT BE PROVEN. The record a stand-down publishes +# removes the task from the watcher's stale and wedge detection, so only a +# proven "no run is active here" may proceed. An absent or unreadable worktree, +# an unreadable branch-scoped status, and a live run the head rule could not +# place all refuse. The optional repo-wide listing cannot weaken a readable +# quiet branch result or make unavailable corroboration a refusal. +# fm_nm_branch_run_verdict carries the query half of the same rule and names +# the one condition that could not prove the branch quiet. +# +# Every kind stand-down accepts is checked the same way, ship and scout alike: +# a scout checked out on a branch can own a run exactly like a ship can, and an +# unchecked hold there would take that run out of supervision. +refuse_stand_down_during_active_run() { + local branch + [ -n "$WT" ] \ + || die "task $ID records no worktree, so whether it owns an active no-mistakes run cannot be checked; refusing to declare a hold that would take a possibly-live run out of supervision" + [ -d "$WT" ] \ + || die "task $ID's worktree $WT is absent or unreadable, so whether it owns an active no-mistakes run cannot be checked; restore the preserved copy before declaring a hold" + git -C "$WT" rev-parse --git-dir >/dev/null 2>&1 \ + || die "task $ID's worktree $WT is not a readable git worktree, so whether it owns an active no-mistakes run cannot be checked; repair it before declaring a hold" + branch=$(git -C "$WT" symbolic-ref --quiet --short HEAD 2>/dev/null || true) + # A worktree with no branch owns no run: a run is keyed by branch, so there + # is nothing for one to hold. That is the scout's ordinary scratch copy, and + # equally a ship between spawn and its worker's first `git checkout -b`, + # which is exactly when an early hold is most likely to be needed. + [ -n "$branch" ] || return 0 + fm_nm_branch_run_verdict "$WT" "$branch" "$NM_CONTROL_TIMEOUT" + case "$FM_NM_BRANCH_RUN_VERDICT" in + active) + die "task $ID owns an active no-mistakes run (${FM_NM_BRANCH_RUN_ID:-unknown}) that still needs a worker at its gates; finish it, or abort it explicitly with 'no-mistakes axi abort --run ${FM_NM_BRANCH_RUN_ID:-}', before standing the worker down" + ;; + quiet) return 0 ;; + unknown) + die "task $ID cannot be stood down because ${FM_NM_BRANCH_RUN_REASON:-the branch-run verdict could not prove it quiet}; refusing to hide a possibly-live run from supervision. Re-run that check yourself, or finish or abort the run, and try again" + ;; + *) die "task $ID's branch-run verdict was invalid; refusing to hide a possibly-live run from supervision" ;; + esac +} + +# Refuse a stand-down while a durable steering instruction is still waiting for +# this task's worker, under the same governing rule: a message nobody has +# acknowledged is work a held worker cannot read, and `fm-send` refuses to add +# another while the endpoint remains deliberately worker-free. +# The instruction is either handled by the worker or explicitly withdrawn by +# the operator - the same acknowledgement move the worker itself makes - and +# neither is something stand-down may decide on the task's behalf. +refuse_stand_down_with_pending_instruction() { + local pending + pending=$(fm_task_inbox_oldest_unhandled "$STATE" "$ID" 2>/dev/null) || return 0 + [ -n "$pending" ] || return 0 + die "task $ID has an unacknowledged instruction waiting at $pending; let the worker handle it, or withdraw it explicitly with 'mv $pending $(fm_task_inbox_handled_dir "$STATE" "$ID")/', before standing the worker down" +} + +# do_repair_worker_state: the one supported reconciliation of a worker-state +# record, so an operator never has to hand-remove one. A valid declaration is +# retained only for a proven dead endpoint. An unprovable record, a live agent, +# or an endpoint that cannot be proved dead clears toward ordinary supervision. +do_repair_worker_state() { + local state outcome + require_worker_lifecycle_backend repair-worker-state + state=$(agent_state 2>/dev/null || true) + [ -n "$state" ] || state=unreadable + outcome=$(fm_worker_state_repair "$STATE" "$ID" "$T" "$state") \ + || die "task $ID's worker-state record could not be removed; check the permissions on $(fm_worker_state_path "$STATE" "$ID")" + case "$outcome" in + cleared-live-worker) + echo "warning: task $ID declared no worker but its endpoint has a live agent; the declaration was cleared and the task is back under ordinary supervision" >&2 + ;; + cleared-invalid) + echo "warning: task $ID had a worker-state record that no longer describes this task and endpoint; it was cleared and the task is back under ordinary supervision" >&2 + ;; + cleared-unproven-endpoint) + echo "warning: task $ID declared no worker but its endpoint could not be proven dead; the declaration was cleared and the task is back under ordinary supervision" >&2 + ;; + esac + printf '%s agent-state=%s' "$outcome" "$state" +} + # --- transactional relaunch ------------------------------------------------- # # The transaction's durable record is state/.control-relaunch, with the @@ -960,12 +1180,16 @@ record_note() { } do_relaunch() { - local exit_result state note_line + local exit_result state note_line worker_state local -a spawn_args require_state_verified_backend relaunch resolve_relaunch_profile + worker_state=$(fm_worker_state_status "$STATE" "$ID" "$T") + [ "$worker_state" != invalid ] \ + || die "task $ID has an invalid worker-state record; refusing to relaunch until it is reconciled with 'fm-control $ID repair-worker-state'" + case "$KIND" in ship|scout) RELAUNCH_BRIEF="$DATA/$ID/brief.md" @@ -1067,6 +1291,14 @@ case "$VERB" in result=$(do_exit) echo "$result $ID harness=$HARNESS backend=$BACKEND endpoint=$T worktree=$WT" ;; + stand-down) + result=$(do_stand_down) + echo "$result $ID harness=$HARNESS backend=$BACKEND endpoint=$T worktree=$WT" + ;; + repair-worker-state) + result=$(do_repair_worker_state) + echo "$result $ID harness=$HARNESS backend=$BACKEND endpoint=$T worktree=$WT" + ;; relaunch) do_relaunch ;; diff --git a/bin/fm-crew-state.sh b/bin/fm-crew-state.sh index 26e5b2d26cb..d40673b097b 100755 --- a/bin/fm-crew-state.sh +++ b/bin/fm-crew-state.sh @@ -175,6 +175,8 @@ STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" . "$SCRIPT_DIR/fm-classify-lib.sh" # shellcheck source=bin/fm-busy-lib.sh . "$SCRIPT_DIR/fm-busy-lib.sh" +# shellcheck source=bin/fm-worker-state-lib.sh +. "$SCRIPT_DIR/fm-worker-state-lib.sh" # shellcheck source=bin/fm-nm-run-lib.sh . "$SCRIPT_DIR/fm-nm-run-lib.sh" # shellcheck source=bin/fm-pr-lib.sh @@ -316,6 +318,82 @@ fi TASK_BACKEND=$(fm_backend_of_meta "$META") BACKEND_TARGET=$(fm_backend_target_of_meta "$META") EXPECTED_LABEL="fm-$ID" + + +# A proven intentional stand-down outranks a HISTORICAL terminal validation +# result: that run describes the last worker incarnation, while this record +# describes the task's current deliberate absence of a worker. It never +# outranks an ACTIVE run, which still owns the branch and still has a gate or a +# step to report - that reporting is the authoritative current state, and +# dropping it would hide actionable work. So the record is resolved here but +# reported only at the two points below where nothing more current exists. +WORKER_LIFECYCLE=$(fm_worker_state_status "$STATE" "$ID" "$BACKEND_TARGET") + +# Report the worker-state verdict, or return 1 to let the caller carry on. +# Recheck the endpoint first so a stale record can never hide a live worker +# that must remain eligible for wedge detection. +# +# Mode `proven-only` reports just the one verdict this record can prove - the +# task really is worker-free - and stays silent otherwise, so a record that +# does NOT describe reality never masks a source that does. Mode `full` also +# surfaces the discrepancies, and is used only where the alternative is a +# guess from a pane read or a stale status log. +emit_worker_state_if_current() { # + local mode=$1 + case "$WORKER_LIFECYCLE" in + stood-down) + case "$(fm_backend_agent_state "$TASK_BACKEND" "$BACKEND_TARGET" 2>/dev/null || true)" in + dead) + # A hold is only healthy while there is no work in flight. A worker- + # state record describes the absence of a worker, never the absence + # of work, so a live run on the preserved branch is reported with its + # details withheld rather than as a healthy park. The other arms below + # are not healthy holds and keep reporting on their own evidence. + if [ "$mode" = full ] && branch_run_verdict_is_active; then + emit working run-step "active run (details withheld)${FM_NM_BRANCH_RUN_ID:+${SEP}run: $FM_NM_BRANCH_RUN_ID}" + fi + emit parked worker-state "worker deliberately stood down" + ;; + # A VANISHED endpoint is not the hold the operator declared. `dead` is + # the declared state: the endpoint is still there, still holds the + # worktree and the uncommitted work, and a relaunch restores the worker + # in place. `missing` means that endpoint is gone, so the hold can no + # longer be resumed where it was declared and something must be done + # about it - reporting it as a healthy park would hide exactly that. + # Only a human declaration ever makes an absent worker healthy here; + # this branch is why no absence is ever inferred to be one. + missing) + [ "$mode" = full ] || return 1 + emit unknown worker-state "stood down but its endpoint is gone: $BACKEND_TARGET (relaunch cannot restore it in place)" + ;; + alive) + [ "$mode" = full ] || return 1 + emit unknown worker-state "record says stood down but endpoint has a live worker" + ;; + *) + [ "$mode" = full ] || return 1 + emit unknown worker-state "stood-down record cannot prove the endpoint remains worker-free" + ;; + esac + ;; + invalid) + [ "$mode" = full ] || return 1 + emit unknown worker-state "invalid worker-state record; reconcile it with 'fm-control $ID repair-worker-state'" + ;; + esac + return 1 +} + +# 0 when this branch provably owns a live no-mistakes run. Consulted only where +# a proven hold would otherwise answer, so ordinary run attribution below +# remains the single owner of run reporting. +branch_run_verdict_is_active() { + FM_NM_BRANCH_RUN_ID= + [ "$KIND" = ship ] && [ -n "$CREW_BRANCH" ] || return 1 + fm_nm_branch_run_verdict "$WT" "$CREW_BRANCH" "$NM_TIMEOUT" + [ "$FM_NM_BRANCH_RUN_VERDICT" = active ] +} + pane_readable() { # case "$TASK_BACKEND" in tmux) tmux display-message -p -t "$1" '#{pane_id}' >/dev/null 2>&1 ;; @@ -1221,6 +1299,14 @@ if [ "$HAVE_RUN" = 1 ]; then ;; esac + # A terminal run is history; an intentional stand-down published after it is + # the newer statement about this task, so it outranks the terminal outcome + # only. Anything the run still reports as live (working, parked at a gate) + # stays authoritative. + case "$RUN_STATE" in + done|failed) emit_worker_state_if_current proven-only || true ;; + esac + [ -z "$SELECTED_RUN_ID" ] || RUN_DETAIL="$RUN_DETAIL${SEP}run: $SELECTED_RUN_ID" emit "$RUN_STATE" run-step "$RUN_DETAIL" fi @@ -1233,6 +1319,11 @@ fi # both classifier-backed backends (tmux and herdr) - and every death-class # verdict reports unknown rather than trusting a possibly-stale status log as # the current state. +# With no run to consult, a worker-state record is the most current statement +# there is about this task - including the discrepancies, whose only remaining +# alternative is a guess from the pane or a stale status log. +emit_worker_state_if_current full || true + [ -n "$BACKEND_TARGET" ] || emit unknown none "no backend target recorded" if ! pane_readable "$BACKEND_TARGET"; then # A failed probe is not itself evidence the pane is gone: the herdr CLI can diff --git a/bin/fm-nm-run-lib.sh b/bin/fm-nm-run-lib.sh index f0e4e83b2ee..3e365ce540b 100644 --- a/bin/fm-nm-run-lib.sh +++ b/bin/fm-nm-run-lib.sh @@ -507,3 +507,181 @@ fm_nm_runs_status_for_worktree() { # [ex printf '%s' "$decided" return 0 } + +# The bounded corroboration window read from `no-mistakes runs`. The listing is +# repo-wide and has neither pagination nor an end-of-list marker, so it can add +# a run this branch owns but can never be asked to prove the absence of one. +fm_nm_runs_limit() { + local n=${FM_CREW_STATE_RUNS_LIMIT:-200} + case "$n" in ''|*[!0-9]*|0) n=200 ;; esac + printf '%s' "$n" +} + +# `no-mistakes` answers a repository it holds no registration for with a +# not-initialized error and no rows at all. A repository with no registration +# owns no run, so that is an answer, not a failure to answer - the same +# reasoning the branchless worktree rests on. firstmate supports whole project +# modes (direct-PR, local-only) that never run `no-mistakes init`. +fm_nm_says_unregistered() { # ... + local response line + for response in "$@"; do + while IFS= read -r line; do + case "$line" in + "repo not initialized (run 'no-mistakes init' first)"|\ + "error: repo not initialized (run 'no-mistakes init' first)") return 0 ;; + esac + done <<< "$response" + done + return 1 +} + +# Bounded `no-mistakes` call in $1 whose stdout and stderr are written into the +# variables named by $2 and $3 rather than stderr being discarded, so a caller +# can inspect the CLI's complete response when it explains a failure on either +# stream. Both variables are assigned in the caller's scope - the call must NOT +# be wrapped in a command +# substitution - and the CLI's exit status is returned, or 1 when no scratch +# file could be opened for the error stream. +fm_nm_run_capturing_stderr() { # + local dir=$1 outvar=$2 errvar=$3 timeout=$4 err='' rc=0 out='' + shift 4 + printf -v "$outvar" '%s' '' + printf -v "$errvar" '%s' '' + if command -v mktemp >/dev/null 2>&1; then + err=$(mktemp "${TMPDIR:-/tmp}/fm-nm-err.XXXXXX" 2>/dev/null) || err= + fi + if [ -z "$err" ]; then + err="${TMPDIR:-/tmp}/fm-nm-err.$$.${RANDOM}${RANDOM}" + ( set -o noclobber; : > "$err" ) 2>/dev/null || return 1 + fi + if out=$(fm_nm_run_bounded "$dir" "$timeout" "$@" 2>"$err"); then rc=0; else rc=$?; fi + printf -v "$outvar" '%s' "$out" + printf -v "$errvar" '%s' "$(<"$err")" + rm -f "$err" 2>/dev/null || : + return "$rc" +} + +# shellcheck disable=SC2034 # FM_NM_BRANCH_RUN_* is this function's documented output tuple. +# `fm_nm_branch_run_verdict` is the one authoritative answer to whether a +# branch owns a live no-mistakes run. It writes `active`, `quiet`, or `unknown` +# to FM_NM_BRANCH_RUN_VERDICT, with one diagnostic reason in +# FM_NM_BRANCH_RUN_REASON when the answer is unknown. +# +# THE QUESTION IS ALWAYS "DOES THIS BRANCH HAVE AN ACTIVE RUN?", AND THE BRANCH +# READ ANSWERS IT. `no-mistakes axi status` reports the repository's ACTIVE run +# and only falls back to the most recent one when nothing is in flight, so a +# successful call that places no non-terminal run on this branch is an answer: +# this branch is quiet, whether the CLI named another branch's run or had +# nothing to report at all. Only a branch read that could not be made - the CLI +# failed or timed out, named an active run without a placeable branch identity, +# or named a live run on this branch whose head cannot be placed against this +# worktree - leaves the question open, and an open question refuses the caller +# that needs it proven. +# +# The repo-wide `runs` listing is CORROBORATION ONLY. The NEWEST row for this +# branch is a second way to establish `active` when it is non-terminal (the +# listing's status column is each run's current status, so it catches a run a +# stale `axi status` answer missed). Only that newest row is read: the listing +# is ordered newest first, so an older live row never displaces the newer +# terminal result that superseded it. Its absence proves nothing: a full window, an unreadable row, a +# failed call, an unregistered repo or an absent CLI must never turn a readable +# quiet branch read into a refusal. That is why widening FM_CREW_STATE_RUNS_LIMIT +# is a reporting nicety rather than a safety setting. +# +# The verdict carries no run detail: it answers the safety question and names +# the active run's id when the branch read placed one. Callers that need run +# detail read it through their own attribution. A direct active answer skips +# the optional listing entirely. +fm_nm_branch_run_verdict() { # [limit] + local wt=$1 branch=$2 timeout=$3 limit=${4:-} + local status_out='' status_rc status_stderr='' status_body='' direct_status='' + local run_branch_raw='' run_branch='' run_head='' + local branch_state=unknown branch_reason='' active_id='' + local inventory row st rest br sha listing_live='' + FM_NM_BRANCH_RUN_VERDICT=unknown + FM_NM_BRANCH_RUN_REASON= + FM_NM_BRANCH_RUN_ID= + case "$limit" in ''|*[!0-9]*|0) limit=$(fm_nm_runs_limit) ;; esac + [ -d "$wt" ] || { + FM_NM_BRANCH_RUN_REASON="worktree '$wt' is not readable, so whether branch '$branch' has a run in flight could not be read" + return 0 + } + [ -n "$branch" ] || { + FM_NM_BRANCH_RUN_VERDICT=quiet + return 0 + } + command -v no-mistakes >/dev/null 2>&1 || { + FM_NM_BRANCH_RUN_VERDICT=quiet + return 0 + } + if fm_nm_run_capturing_stderr "$wt" status_out status_stderr "$timeout" axi status; then status_rc=0; else status_rc=$?; fi + if [ "$status_rc" = 0 ]; then + branch_state=quiet + status_body=$(fm_nm_trim "$status_out") + if [ -n "$status_body" ]; then + direct_status=$(fm_nm_strip_quotes "$(fm_nm_field "$status_out" status)") + run_branch_raw=$(fm_nm_trim "$(fm_nm_field "$status_out" branch)") + run_branch=$(fm_nm_strip_quotes "$run_branch_raw") + run_head=$(fm_nm_strip_quotes "$(fm_nm_field "$status_out" head)") + if [ -z "$direct_status" ]; then + branch_state=unknown + branch_reason="'no-mistakes axi status' returned a non-empty result without a readable run status for branch $branch" + elif fm_nm_run_is_active "$status_out"; then + case "$run_branch_raw" in + \"*\") ;; + *\"*) run_branch= ;; + esac + if [ -z "$run_branch" ] \ + || ! git check-ref-format --branch "$run_branch" >/dev/null 2>&1; then + branch_state=unknown + branch_reason="the active run 'no-mistakes axi status' reports has no placeable branch identity, so whether it belongs to branch $branch cannot be proved" + elif [ "$run_branch" = "$branch" ]; then + if fm_nm_head_matches_worktree "$wt" "$run_head" \ + || fm_nm_run_is_pipeline_owned_active "$status_out"; then + branch_state=active + active_id=$(fm_nm_strip_quotes "$(fm_nm_field "$status_out" id)") + else + branch_state=unknown + branch_reason="the run 'no-mistakes axi status' reports on branch $branch (head ${run_head:-unknown}) cannot be placed against this worktree's HEAD" + fi + fi + fi + fi + elif [ "$status_rc" != 0 ] \ + && fm_nm_says_unregistered "$status_out" "$status_stderr"; then + branch_state=quiet + else + branch_reason="'no-mistakes axi status' could not answer whether branch $branch has a run in flight" + fi + if [ "$branch_state" != active ] \ + && inventory=$(fm_nm_run_checked "$wt" "$timeout" runs --limit "$limit"); then + while IFS= read -r row; do + row=$(fm_nm_trim "$row") + [ -n "$row" ] || continue + case "$row" in *" "*) ;; *) continue ;; esac + st=${row%% *} + rest=$(fm_nm_trim "${row#* }") + [ -n "$rest" ] || continue + br=${rest%% *} + rest=$(fm_nm_trim "${rest#* }") + sha=${rest%% *} + { [ -n "$br" ] && [ -n "$sha" ]; } || continue + [ "$br" = "$branch" ] || continue + case "$st" in + completed|failed|cancelled) ;; + *) listing_live=$st ;; + esac + break + done <<< "$inventory" + fi + if [ "$branch_state" = active ]; then + FM_NM_BRANCH_RUN_VERDICT=active + FM_NM_BRANCH_RUN_ID=$active_id + elif [ -n "$listing_live" ]; then + FM_NM_BRANCH_RUN_VERDICT=active + elif [ "$branch_state" = quiet ]; then + FM_NM_BRANCH_RUN_VERDICT=quiet + else + FM_NM_BRANCH_RUN_REASON=$branch_reason + fi +} diff --git a/bin/fm-send.sh b/bin/fm-send.sh index cdd4ddcb95d..90918a6a2c5 100755 --- a/bin/fm-send.sh +++ b/bin/fm-send.sh @@ -27,7 +27,8 @@ # durably sent (recorded); nonzero = nothing was confirmed delivered and a # resend is appropriate (unresolvable target, an endpoint that cannot be # locked and revalidated or that retired or changed, an unwritable record, a -# failed or lost remote transport) or a decision-close append failed after +# failed or lost remote transport, or a task deliberately recorded as having +# no worker) or a decision-close append failed after # delivery (the error then carries the exact manual close). The remote enqueue # is idempotent: the remote leg deduplicates an exact re-run of the same # request onto the existing record (bin/fm-task-inbox-lib.sh), so after a lost @@ -242,6 +243,8 @@ fi . "$SCRIPT_DIR/fm-backend.sh" # shellcheck source=bin/fm-control-lib.sh . "$SCRIPT_DIR/fm-control-lib.sh" +# shellcheck source=bin/fm-worker-state-lib.sh +. "$SCRIPT_DIR/fm-worker-state-lib.sh" # shellcheck source=bin/fm-marker-lib.sh . "$SCRIPT_DIR/fm-marker-lib.sh" # shellcheck source=bin/fm-pending-reply-lib.sh @@ -443,6 +446,23 @@ fm_send_resolve_target "$RAW_TARGET" || exit 1 T=$RESOLVED_TARGET shift +# A held task with a proven stood-down worker in its recorded endpoint has no +# receiver for either inbox or typed-plane input. Refuse instead of leaving an +# unread durable instruction. Missing or unprovable endpoints remain under +# ordinary supervision, and a stale record never blocks a live endpoint. +if [ -n "$TARGET_META" ] && [ "$TARGET_BACKEND" != remote ]; then + TARGET_WORKER_ID=$(fm_send_id_from_meta "$TARGET_META") + if [ "$(fm_worker_state_status "$STATE" "$TARGET_WORKER_ID" "$T")" = stood-down ]; then + TARGET_WORKER_AGENT_STATE=$(fm_backend_agent_state "$TARGET_BACKEND" "$T" 2>/dev/null || true) + case "$TARGET_WORKER_AGENT_STATE" in + dead) + echo "error: task $TARGET_WORKER_ID deliberately has no worker; relaunch it before sending an instruction" >&2 + exit 1 + ;; + esac + fi +fi + # Supervision lease guard: a steer is overlap territory between the two Pi # supervision actors, so refuse while the OTHER actor holds this task's live # lease. A home with no supervision branch has no lease files and passes diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index fcee5a152b3..b3a03e92d2e 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -621,6 +621,8 @@ fm_backlog_directory_present "$STATE" "state directory" || { . "$SCRIPT_DIR/fm-backend.sh" # shellcheck source=bin/fm-control-lib.sh . "$SCRIPT_DIR/fm-control-lib.sh" +# shellcheck source=bin/fm-worker-state-lib.sh +. "$SCRIPT_DIR/fm-worker-state-lib.sh" # shellcheck source=bin/fm-gate-refuse-lib.sh . "$SCRIPT_DIR/fm-gate-refuse-lib.sh" # shellcheck source=bin/fm-busy-lib.sh @@ -1210,6 +1212,15 @@ RELAUNCH_REPLACEMENT_BUSY_GEN= RELAUNCH_REPLACEMENT_HARNESS= RELAUNCH_REPLACEMENT_STATE= RELAUNCH_REPLACEMENT_WT= +# The intentional no-worker declaration this relaunch cleared, if any, so an +# abort BEFORE the replacement's metadata is published can put the task back +# in the state the operator left it in rather than in an undeclared no-worker +# limbo the watcher would false-alarm on. Restored under exactly the same +# pre-publication gate as the rest of the replacement rollback: once the +# replacement is published an agent is about to start, and a stand-down +# declaration must not survive that. +RELAUNCH_REPLACEMENT_WORKER_STATE= +RELAUNCH_REPLACEMENT_ENDPOINT= CONFIG_INHERIT_LOCK= CONFIG_INHERIT_LOCK_HELD=0 GIT_HOOKS_DIR= @@ -1254,6 +1265,13 @@ spawn_abort_cleanup() { fi if [ "$RELAUNCH_REPLACEMENT_PENDING" = 1 ]; then RELAUNCH_REPLACEMENT_PENDING=0 + if [ -n "$RELAUNCH_REPLACEMENT_WORKER_STATE" ]; then + if ! fm_worker_state_write "$RELAUNCH_REPLACEMENT_STATE" "$ID" \ + "$RELAUNCH_REPLACEMENT_ENDPOINT" "$RELAUNCH_REPLACEMENT_WORKER_STATE"; then + echo "warning: could not restore the intentional no-worker record for $ID after an aborted relaunch; it is under ordinary supervision until 'fm-control $ID stand-down' re-declares the hold" >&2 + fi + RELAUNCH_REPLACEMENT_WORKER_STATE= + fi if ! clear_relaunch_harness_wiring \ "$RELAUNCH_REPLACEMENT_HARNESS" \ "$RELAUNCH_REPLACEMENT_WT" \ @@ -4425,6 +4443,35 @@ exclude_path() { grep -qxF "$rel" "$EXCL" 2>/dev/null || echo "$rel" >>"$EXCL" } if [ "$RELAUNCH" -eq 1 ]; then + # A replacement worker makes a deliberate no-worker declaration stale before + # it can become live, so the record is resolved as part of arming the + # replacement. A record whose meaning cannot be proved refuses HERE, while + # the task still has its prior incarnation's wiring: this exit rolls nothing + # back, so nothing may have been retired yet. + RELAUNCH_PRIOR_WORKER_STATE=$(fm_worker_state_status "$STATE_REAL" "$ID" "$T") + [ "$RELAUNCH_PRIOR_WORKER_STATE" != invalid ] || { + echo "error: task $ID has an invalid worker-state record; refusing to relaunch until it is reconciled with 'fm-control $ID repair-worker-state'" >&2 + exit 1 + } + # Remember what the declaration was before anything is retired: an abort + # before the replacement is published must hand the task back exactly as + # declared, not leave it worker-free AND undeclared. + RELAUNCH_REPLACEMENT_HARNESS=$HARNESS + RELAUNCH_REPLACEMENT_STATE=$STATE_REAL + RELAUNCH_REPLACEMENT_WT=$WT + RELAUNCH_REPLACEMENT_ENDPOINT=$T + case "$RELAUNCH_PRIOR_WORKER_STATE" in + standing-down|stood-down) RELAUNCH_REPLACEMENT_WORKER_STATE=$RELAUNCH_PRIOR_WORKER_STATE ;; + *) RELAUNCH_REPLACEMENT_WORKER_STATE= ;; + esac + # Resolving the record comes FIRST, for the same reason the invalid-record + # refusal does: a removal that fails here has retired nothing, so the task + # keeps both its declaration and its prior incarnation's wiring. + fm_worker_state_clear "$STATE_REAL" "$ID" "$T" || { + echo "error: task $ID's worker-state record could not be cleared for the replacement; reconcile it with 'fm-control $ID repair-worker-state' and relaunch again" >&2 + exit 1 + } + RELAUNCH_REPLACEMENT_PENDING=1 # Retire the previous incarnation's per-task harness wiring before arming the # new one. Without this, a harness switch would leave the old adapter's hook # files and turn-end token registry entries behind, and even a same-harness @@ -4434,10 +4481,6 @@ if [ "$RELAUNCH" -eq 1 ]; then echo "error: could not retire $RELAUNCH_PRIOR_HARNESS wiring for task $ID; refusing to arm the replacement" >&2 exit 1 } - RELAUNCH_REPLACEMENT_PENDING=1 - RELAUNCH_REPLACEMENT_HARNESS=$HARNESS - RELAUNCH_REPLACEMENT_STATE=$STATE_REAL - RELAUNCH_REPLACEMENT_WT=$WT fi if [ "$KIND" != secondmate ]; then # Arm the semantic busy-state contract (bin/fm-busy-lib.sh) for every diff --git a/bin/fm-teardown.sh b/bin/fm-teardown.sh index 8b9603d7e2d..4307314076b 100755 --- a/bin/fm-teardown.sh +++ b/bin/fm-teardown.sh @@ -3795,6 +3795,7 @@ retire_busy_state "$STATE" "$ID" "$BUSY_GEN" || exit 1 status_retire_presentation_task "$STATE" "$ID" || exit 1 fm_wake_queue_prune_task "$STATE" "$ID" "$T" 2>/dev/null || true rm -f "$STATE/$ID.turn-ended" "$STATE/$ID.progress" \ + "$STATE/$ID.worker-state" \ "$(fm_wake_signal_seen_path "$STATE" "$STATE/$ID.turn-ended")" \ "$STATE/$ID.pi-ext.ts" "$STATE/$ID.omp-ext.ts" "$STATE/$ID.grok-turnend-token" \ "$STATE/$ID.kimi-turnend-token" "$STATE/$ID.muse-session" \ diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index dcd1f7c4933..eb88059bacf 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -215,6 +215,8 @@ WATCH_HOME_EXISTED=0 . "$SCRIPT_DIR/fm-pending-reply-lib.sh" # shellcheck source=bin/fm-busy-lib.sh . "$SCRIPT_DIR/fm-busy-lib.sh" +# shellcheck source=bin/fm-worker-state-lib.sh +. "$SCRIPT_DIR/fm-worker-state-lib.sh" # Steering-inbox loss detection: bin/fm-task-inbox-lib.sh owns the record, # doorbell, re-ring ladder, and unavailable-endpoint contracts; this watcher # supplies their live endpoint and busy checks plus wake emission @@ -826,6 +828,28 @@ recorded_windows() { done } +# 0 when is a task deliberately left with no worker: an exact +# worker-state record AND a recovery-grade classifier that still proves the +# endpoint has no agent. A live replacement behind a stale record is never +# exempt - it re-enters stale and wedge detection on the same poll. +# +# The exemption is deliberately scoped to the pane-stale and wedge work at its +# one call site, NOT to every per-window check. The fast push-event path needs +# no exemption of its own: it only ever sees push-capable windows, and a hold is +# tmux-only, so a stood-down window never reaches it. The steering-inbox re-ring +# ladder keeps running for a held task, because the two guards that keep an +# inbox empty at stand-down time (fm-control's pending-instruction refusal and +# fm-send's refusal to enqueue for a proven worker-free task) take different +# locks and so cannot exclude a steer that lands on the record's other side. +# Supervising that message costs one ladder check and is the only thing that +# surfaces it before a relaunch. +window_is_stood_down() { # + local w=$1 task=$2 + [ -n "$task" ] || return 1 + [ "$(fm_worker_state_status "$STATE" "$task" "$w")" = stood-down ] || return 1 + [ "$(fm_backend_agent_state "$(window_backend "$w")" "$w" 2>/dev/null || true)" = dead ] +} + # Print the oldest structurally valid ACTIONABLE row in a local secondmate's # foreign queue. A stale recheck that explicitly identifies itself as a declared # external-wait pause is not evidence that the mate's wake loop is stuck: the @@ -3001,6 +3025,7 @@ EOF # Steering-inbox loss detection runs before the secondmate stale # exemption below, because a mate's steers land in an inbox too. [ -z "$task" ] || inbox_steer_check "$w" "$task" + window_is_stood_down "$w" "$task" && continue key=$(window_key "$w") last=$(status_declared_wait_line "$STATE/$task.status") if ! status_is_paused_or_captain_held "$last" && [ -e "$STATE/.paused-$key" ]; then diff --git a/bin/fm-worker-state-lib.sh b/bin/fm-worker-state-lib.sh new file mode 100644 index 00000000000..b4cd6a9b345 --- /dev/null +++ b/bin/fm-worker-state-lib.sh @@ -0,0 +1,153 @@ +#!/usr/bin/env bash +# fm-worker-state-lib.sh - the one durable representation of an intentionally +# worker-free task. +# +# A task normally retains its endpoint metadata after fm-control exit so a +# later relaunch can reuse the same worktree and endpoint. That is not proof +# that an absent agent is intentional: an ordinary exit can be a crash. Only +# fm-control stand-down writes state/.worker-state, and it writes +# `stood-down` only after the backend's recovery-grade classifier proves the +# agent is dead. The watcher skips a matching record only while that same +# classifier still sees the endpoint as dead; a live agent always remains in +# stale and wedge detection. +# +# Record format, written atomically with mode 0600: +# schema=1 +# task_id= +# endpoint= +# state=standing-down|stood-down +# +# `standing-down` records the short transition before the exit is proved. It +# is never a suppression state, so an interrupted control action fails toward +# ordinary supervision. `stood-down` is cleared by fm-spawn --relaunch before +# it prepares the replacement, and restored if that relaunch aborts before the +# replacement's metadata is published, so a failed relaunch returns the task to +# the declaration it started from. Teardown removes the record with the task's +# other endpoint-local runtime state. +# +# `fm-control repair-worker-state` is the only supported way to reconcile +# a record that no longer describes reality (fm_worker_state_repair below); +# nothing here is meant to be hand-edited or hand-removed. +# +# This is a sidecar rather than a state/.meta field. Endpoint metadata is +# the relaunch identity and has independent transactional writers, so mixing a +# worker-lifecycle declaration into it would make an innocent hold capable of +# corrupting or racing that identity. + +fm_worker_state_path() { # + printf '%s/%s.worker-state' "$1" "$2" +} + +# fm_worker_state_status +# Prints one of active, standing-down, stood-down, or invalid. A record is +# valid only when every schema key appears once and the task/endpoint binding +# still matches the caller's current metadata-derived identity. +fm_worker_state_status() { + local state_dir=$1 id=$2 endpoint=$3 record key count schema task bound state + record=$(fm_worker_state_path "$state_dir" "$id") + { [ -e "$record" ] || [ -L "$record" ]; } || { printf 'active'; return 0; } + [ -f "$record" ] && [ ! -L "$record" ] || { printf 'invalid'; return 0; } + for key in schema task_id endpoint state; do + count=$(awk -F= -v key="$key" '$1 == key { n += 1 } END { print n + 0 }' "$record" 2>/dev/null) || { + printf 'invalid' + return 0 + } + [ "$count" = 1 ] || { printf 'invalid'; return 0; } + done + schema=$(awk -F= '$1 == "schema" { sub(/^[^=]*=/, ""); print; exit }' "$record" 2>/dev/null) || schema= + task=$(awk -F= '$1 == "task_id" { sub(/^[^=]*=/, ""); print; exit }' "$record" 2>/dev/null) || task= + bound=$(awk -F= '$1 == "endpoint" { sub(/^[^=]*=/, ""); print; exit }' "$record" 2>/dev/null) || bound= + state=$(awk -F= '$1 == "state" { sub(/^[^=]*=/, ""); print; exit }' "$record" 2>/dev/null) || state= + [ "$schema" = 1 ] && [ "$task" = "$id" ] && [ "$bound" = "$endpoint" ] || { + printf 'invalid' + return 0 + } + case "$state" in + standing-down|stood-down) printf '%s' "$state" ;; + *) printf 'invalid' ;; + esac +} + +# fm_worker_state_write +# The caller owns the task lifecycle lock. Refuse to replace malformed or +# symlinked state, because overwriting a record whose meaning cannot be proved +# would turn a failed control action into a false intentional stand-down. +fm_worker_state_write() { + local state_dir=$1 id=$2 endpoint=$3 wanted=$4 record tmp prior + case "$id" in ''|*[!A-Za-z0-9._-]*) return 1 ;; esac + case "$endpoint" in ''|*$'\n'*) return 1 ;; esac + case "$wanted" in standing-down|stood-down) ;; *) return 1 ;; esac + record=$(fm_worker_state_path "$state_dir" "$id") + if [ -e "$record" ] || [ -L "$record" ]; then + [ -f "$record" ] && [ ! -L "$record" ] || return 1 + prior=$(fm_worker_state_status "$state_dir" "$id" "$endpoint") + [ "$prior" != invalid ] || return 1 + fi + tmp="$state_dir/.${id}.worker-state.${BASHPID:-$$}.${RANDOM}" + if ! ( + umask 077 + { + printf 'schema=1\n' + printf 'task_id=%s\n' "$id" + printf 'endpoint=%s\n' "$endpoint" + printf 'state=%s\n' "$wanted" + } > "$tmp" + ); then + rm -f -- "$tmp" + return 1 + fi + if ! mv -f -- "$tmp" "$record"; then + rm -f -- "$tmp" + return 1 + fi +} + +# fm_worker_state_repair +# The one supported reconciliation of a record against observed reality, so no +# operator ever has to hand-edit or hand-remove one. A valid declaration is +# retained only while the endpoint is positively classified as dead. A record +# whose meaning cannot be proved, a declaration contradicted by a live agent, +# or an endpoint whose absence cannot be proved is removed toward ordinary +# supervision. Idempotent: a second call on an already-reconciled task is a +# no-op. +# Prints the outcome word; returns nonzero only when a removal failed. +fm_worker_state_repair() { # + local state_dir=$1 id=$2 endpoint=$3 observed=$4 record status + record=$(fm_worker_state_path "$state_dir" "$id") + { [ -e "$record" ] || [ -L "$record" ]; } || { printf 'no-record'; return 0; } + status=$(fm_worker_state_status "$state_dir" "$id" "$endpoint") + case "$status" in + invalid) + rm -f -- "$record" || return 1 + printf 'cleared-invalid' + ;; + standing-down|stood-down) + case "$observed" in + dead) printf 'intact' ;; + alive) + rm -f -- "$record" || return 1 + printf 'cleared-live-worker' + ;; + *) + rm -f -- "$record" || return 1 + printf 'cleared-unproven-endpoint' + ;; + esac + ;; + *) printf 'intact' ;; + esac +} + +# fm_worker_state_clear +# A relaunch can clear only a valid state record for its own exact endpoint. +# An unprovable record is refused here rather than silently discarded, because +# only the repair path above may resolve one, and only against observed +# reality. +fm_worker_state_clear() { + local state_dir=$1 id=$2 endpoint=$3 record status + record=$(fm_worker_state_path "$state_dir" "$id") + { [ -e "$record" ] || [ -L "$record" ]; } || return 0 + status=$(fm_worker_state_status "$state_dir" "$id" "$endpoint") + [ "$status" != invalid ] || return 1 + rm -f -- "$record" +} diff --git a/docs/agent-control.md b/docs/agent-control.md index c949a13ba96..56dc804ca02 100644 --- a/docs/agent-control.md +++ b/docs/agent-control.md @@ -15,7 +15,7 @@ The failure repeated across harnesses and homes, and the workaround (remember to `bin/fm-control-lib.sh` is the single executable owner of three capability tables, which have no side effects, so they can be read as a contract: -- The **verb allowlist**: `interrupt`, `exit`, `relaunch`. +- The **verb allowlist**: `interrupt`, `exit`, `stand-down`, `repair-worker-state`, `relaunch`. There is no arbitrary-text and no generic raw-key entry point. A caller either names an allowlisted verb or is refused. - **Per-harness mechanics**: the key that cancels a running turn, how many times it must be delivered, whether the composer needs clearing afterwards, the command that exits the agent, and which task kinds the adapter is verified to run. @@ -34,6 +34,8 @@ A recorded `harness=` is not always an exact adapter name: a task launched from | --- | --- | --- | | `interrupt` | Deliver the harness's verified interrupt sequence while leaving the agent running. | Delivery succeeds while the endpoint still exists and the agent is still alive where the backend can classify that; cancellation is confirmed only from an adapter-owned acknowledgement and otherwise reports `cancel=unconfirmed`. | | `exit` | Stop the agent, preserving the endpoint, the worktree, and every uncommitted change. | The backend's recovery-grade classifier reports the agent gone. Already-stopped is idempotent success. An endpoint reading `missing` goes through the same [absence proof](#reclaiming-a-task-whose-endpoint-is-gone) the reclaim uses before anything is claimed about it, and only Herdr can supply one: proven gone reports `endpoint-gone` (the agent went with it, and the endpoint this verb normally preserves did not survive), a pane that turns out to be there and idle is the ordinary `already-stopped`, one whose agent is back takes the ordinary interrupt-then-exit path. A tmux `missing` always refuses rather than claim a stop it cannot see. | +| `stand-down` | Stop a held ship or scout task and record that it deliberately has no worker. | The control plane first proves the agent is gone, then writes an exact worker-state record bound to the task and endpoint. A live worker at a stale record stays in ordinary stale and wedge detection. | +| `repair-worker-state` | Reconcile a worker-state record against what the endpoint really shows. | A valid record is retained only for a proven dead endpoint. An unprovable record, a live agent, or a missing or unreadable endpoint clears the declaration toward ordinary supervision. Repeat runs are no-ops. | | `relaunch` | Replace the running agent with a new one in the same worktree - and the same endpoint whenever that endpoint still exists - on the exact recorded adapter or an explicitly chosen harness, model, and effort. | The new agent is alive on the endpoint the task's record now names, and that record names the harness that is actually running. | An exit that delivers lifecycle input but cannot prove the agent stopped fails with `exit=unconfirmed`, reports the observed agent state and any interrupt cancellation claim, and never claims that nothing changed. @@ -55,6 +57,39 @@ The clear is refused before anything is sent when the recorded backend cannot de `exit` stops an agent and preserves everything else. Removing a worktree, closing an endpoint, or discarding work stays with [`bin/fm-teardown.sh`](../bin/fm-teardown.sh), which owns the landed-work test. +`stand-down` is the explicit companion for a held ship or scout task on the tmux backend that does not need a worker for a while. +Both kinds are checked for an in-flight run the same way, and a worktree with no branch at all - a scout's scratch copy, or a ship between spawn and its worker's first `git checkout -b` - owns no run to be held back by, because a run is keyed by branch. +It preserves the worktree, branch, commits, endpoint, and uncommitted work exactly as `exit` does, then writes `state/.worker-state` only after proving the worker is gone. +Its short `standing-down` transition never suppresses monitoring, and the completed `stood-down` record is ignored when the endpoint has a live worker. +`fm-spawn --relaunch` clears a valid record immediately before preparing the replacement, so the same preserved worktree resumes normally; if that relaunch aborts before the replacement's metadata is published, the record is restored and the task returns to the hold it was in. +`fm-send` refuses new input while the record and endpoint both prove that no worker is present, so a held task cannot accumulate an instruction that nobody can read. + +Three things a stand-down deliberately cannot do. +It cannot run while the task owns an in-flight no-mistakes run: that run owns the branch and needs a worker at its gates, so finish it or abort it yourself (`no-mistakes axi abort --run `) first - stand-down never cancels a run for you. +The branch-run verdict in [`bin/fm-nm-run-lib.sh`](../bin/fm-nm-run-lib.sh) is the one owner of that question, and it always asks the same thing: does THIS branch have a run in flight? +The branch read - `no-mistakes axi status`, which reports the repository's active run and falls back to the most recent one only when nothing is in flight - answers it: a non-terminal run on this branch is `active`, and a readable answer that puts no non-terminal run on this branch is `quiet`. +Only a branch read that could not be interpreted leaves the question open: the CLI failed or timed out, returned a non-empty malformed status, named an active run without a placeable branch identity, or named a live run on this branch whose head cannot be placed against the local HEAD (the head rule exists to reject a historical run on a reused branch, and a run that is still going owns the branch however far local work has advanced past the commit it started on). +An open question is refused, and the refusal names what could not be read - as is an absent or unreadable worktree. +The repo-wide `no-mistakes runs` listing is corroboration only: the newest row for this branch is a second way to reach `active` when it is non-terminal, because the listing's status column is each run's current status and catches a run a stale `axi status` answer missed. +Only that newest row counts - the listing is ordered newest first, so an older live row never displaces the newer terminal run that superseded it. +Its silence proves nothing and never refuses a hold on its own - an exactly full window (the steady state of any mature repository), an unparsable row, a failed call, a repository the CLI holds no registration for, and a home without `no-mistakes` installed at all each simply add no run, so `FM_CREW_STATE_RUNS_LIMIT` is a reporting nicety rather than a safety setting. +A finished run is history either way: a terminal run whose head never reached this worktree still answers the only question the hold depends on, so it never blocks one. +It cannot run while an unacknowledged steering instruction is still waiting in `state/.inbox/`, because a held task's worker cannot read it and `fm-send` refuses to add another: the refusal names the record, and the worker handles it or the operator withdraws it with the same `mv state/.inbox/handled/` acknowledgement the worker would make. +And it cannot turn an agent that is merely already gone into a deliberate hold, because deadness is exactly the ambiguity the record exists to resolve. +To declare an ordinary prior `exit` intentional, first declare the hold the way both supervisors already read it - append a `paused: ` (or `captain-held: `) line to `state/.status` - then run `stand-down`. +That declaration is reversible by the next ordinary status append. + +`repair-worker-state` is the only supported way to reconcile a record; nothing under `state/` is meant to be hand-edited. +Reality wins, and any change moves only toward supervision: a record that no longer describes this task and endpoint, or one a live agent contradicts, is cleared and the discrepancy is reported on stderr. +A valid declaration is preserved while its exact endpoint is proven dead, but a dead endpoint alone never lets repair create a declaration. +Repair can therefore retain an established hold or return a task to ordinary monitoring, but never infer a new hold from worker absence. + +`fm-crew-state` reports a proven stand-down as `state: parked · source: worker-state`, but only where nothing more current exists: an attributed run that is still live keeps run-step authority, so a run parked at a gate reports its own step and findings rather than falling through to the hold. +A terminal run is history, so the record outranks it; [`bin/fm-nm-run-lib.sh`](../bin/fm-nm-run-lib.sh) owns which run is attributed in the first place. +Where no run was attributed but the preserved branch still provably owns a live one, the hold is reported as `working` with `active run (details withheld)` rather than as a healthy park, because the record describes the absence of a worker and never the absence of work. +It is also only ever a park while the recorded endpoint is still there and merely has no agent. +An endpoint that has vanished reports `unknown` and names the lost endpoint, because the declared hold - worktree, work, and an in-place relaunch - can no longer be resumed where it was declared. + **`resume` is not a verb.** It is not deterministic across the verified adapters: codex, grok, gemini, and devin resume only from a session id printed at exit, opencode continues the most recent session for the cwd, and claude, pi, pi-signed, omp, kimi, and agy have no verified general pane-resume contract. `relaunch` uses the brief on disk - not a harness-private session - as the durable instruction when the backend can prove the old agent stopped and the composer is empty; Devin on Herdr currently fails that composer check and refuses. @@ -62,7 +97,7 @@ A relaunch does take one session reference when the endpoint's own runtime recor ## Transactional relaunch -`relaunch` is the only verb that changes durable records, so it runs as a transaction with a journal at `state/.control-relaunch`, the prior record preserved beside it, and a ship or scout's prior instructions preserved when a progress note is appended. +`relaunch` changes task metadata and, for a ship or scout, its instructions, so it runs as a transaction with a journal at `state/.control-relaunch`, the prior record preserved beside it, and the prior instructions preserved when a progress note is appended. 1. **Resolve the profile.** An explicit `--harness`, `--model`, or `--effort` wins. @@ -73,6 +108,7 @@ A relaunch does take one session reference when the endpoint's own runtime recor A Claude or Pi replacement must also pass the home's [worker account pin](configuration.md#worker-account-pin-configclaude-account-configpi-account), so a pin that no longer resolves or is signed out refuses before the old agent stops. 2. **Safe checkpoint.** The recorded worktree must exist and be a worktree root; its head and dirty state are recorded. + Any worker-state record must still bind this task and endpoint; an invalid record refuses before the note is appended or the old agent is stopped. For a `kind=secondmate` task, the home's identity marker must match and its child records must be readable, so a relaunch can never strand child work behind an unreadable home. A secondmate's own crewmates run in their own endpoints and outlive its relaunch; the relaunched secondmate reconciles them from its home's durable records at startup. 3. **Record the note.** @@ -162,8 +198,10 @@ The worktree and the task's records are unaffected either way. Muse is a crewmate and scout adapter only, so relaunching a secondmate onto it refuses while its agent is still up rather than leaving that secondmate with no agent when the launch owner refuses. - A backend that cannot deliver the harness's interrupt key, or the composer clear that key needs, is refused rather than sent a different key. Orca's terminal API exposes only an interrupt and an Enter, so it can deliver neither Escape nor Ctrl+U. -- `exit` and `relaunch` require a backend with a recovery-grade agent-state classifier - tmux and herdr - because without one the "the agent stopped" postcondition cannot be proven. +- `exit` and `relaunch` require a backend with a recovery-grade agent-state classifier - tmux or herdr - because without one the "the agent stopped" postcondition cannot be proven. zellij, orca, and cmux are refused rather than reported as successful blind. +- The worker-state verbs `stand-down` and `repair-worker-state` are supported only on tmux. + They do not drive Herdr lifecycle behaviour. - An ambiguous or unreadable endpoint state refuses. Only a positively classified state acts. - `exit`'s composer-empty check, above, is itself a fail-closed boundary that `relaunch` inherits by stopping the old agent through `exit`. @@ -188,6 +226,7 @@ The empirical basis for each adapter's value is the `harness-adapters` skill's v ## Verification -- `tests/fm-control.test.sh` - the adapter contract for its verified-harness lane (adapters outside the lane pin their control mechanics in their own harness suites), the backend capability matrix, exact-id scoping, the closed verb list, the busy, idle, dead, and idempotent lifecycle cases, and marker non-regression, all against a stubbed session provider. -- `tests/fm-control-relaunch.test.sh` - the relaunch transaction: identity preservation, harness switching, the progress note, checkpoint refusals, rollback after a failed launch, and the endpoint-absence proof both verbs share - the Herdr reclaim of a destroyed endpoint, and tmux refusing one it cannot prove absent. +- `tests/fm-control.test.sh` - the adapter contract for its verified-harness lane (adapters outside the lane pin their control mechanics in their own harness suites), the backend capability matrix, exact-id scoping, the closed verb list, the busy, idle, dead, idempotent, and deliberate stand-down lifecycle cases, every stand-down refusal that carries the burden of proof (active run, unplaceable run, unanswerable check, unreadable worktree, pending instruction), and marker non-regression, all against a stubbed session provider. +- `tests/fm-crew-state.test.sh` - includes the absence-as-healthy counterfactual: a stood-down record whose endpoint has vanished must report `unknown` and name the lost endpoint, so the test fails the moment absence is presented as a healthy hold, and an absent worker with no declaration at all is still reported as a problem. +- `tests/fm-control-relaunch.test.sh` - the relaunch transaction: identity preservation, a restart from a deliberately stood-down worker, harness switching, the progress note, checkpoint refusals, rollback after a failed launch, and the endpoint-absence proof both verbs share - the Herdr reclaim of a destroyed endpoint, and tmux refusing one it cannot prove absent. - `tests/fm-control-herdr-smoke.test.sh` - the second state-verified backend against the real herdr binary, on an isolated throwaway lab session. diff --git a/docs/architecture.md b/docs/architecture.md index a8754ae3bee..8011772f1af 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -92,6 +92,8 @@ If two metadata records derive the same per-window marker key, including two rec A `kind=secondmate` task's status stream doubles as its parent-directed reply channel, so its lines new since the last classification are read before busy evidence counts: a decision, blocker, terminal outcome, `note:`, correlation-marked line, or unknown verb always surfaces, while unmarked routine `working:` and `paused:` progress is absorbed only by the same provably-working proof an ordinary crewmate gets. Its bare turn-ended signal is absorbed only by the ordinary authoritative working proof because an active secondmate does not enter the staleness backbone that would resurface deferred pane-churn evidence. A crew that declares `paused:` for a known external wait, or carries a verified `captain-held` transfer, is separately absorbed while idle and re-surfaced only on the longer pause cadence, rather than being treated as a possible wedge, except that a captain-held transfer is not rechecked while the away-posture record exists. +An exact `worker-state` record written by `fm-control stand-down` excludes a dead endpoint from pane-stale polling and wedge escalation while the held task has intentionally no worker, but a live endpoint at that record immediately returns to ordinary stale and wedge detection. +The exemption stops there: the steering-inbox re-ring ladder keeps running for a held task, so a steer that lands on the other side of the record - `stand-down` refuses while an instruction is unacknowledged and `fm-send` refuses to enqueue one afterwards, but those two guards take different locks - is still surfaced rather than waiting silently for a relaunch. For an ordinary crew that has stopped, the normal-mode watcher first surfaces one stale wake, then applies that same cadence to an unchanged `paused:` or durable `captain-held` endpoint while attended; the pause classification itself is recovered only when the backend confidently reports its agent dead. Live or inconclusive liveness remains fail-open at that initial surface, so a worker genuinely waiting on a decision is never silenced. Its later sights are still held to that same bounded cadence rather than re-alarming on every pane-hash change, because the throttle is keyed to the declaration and not to the pane an idle parked worker keeps ticking. @@ -226,7 +228,7 @@ Its local-only typed plane - harness-native invocations and explicit backend tar Text for a worker to read and commands that drive a worker's process are separate planes. `fm-send.sh` is the data plane and always routing-marks a `kind=secondmate` target, which is right for a message and wrong for a lifecycle command, because a marked exit command arrives as chat the agent reasons about instead of executing. -`bin/fm-control.sh` is the control plane: an allowlisted `interrupt`, `exit`, and transactional `relaunch` addressed to an exact task id, with per-harness mechanics owned by `bin/fm-control-lib.sh`, a verified postcondition per verb, and no arbitrary-text or raw-key entry point. +`bin/fm-control.sh` is the control plane: a closed allowlist of lifecycle and worker-state verbs addressed to an exact task id, with per-harness mechanics owned by `bin/fm-control-lib.sh`, a verified postcondition per verb, and no arbitrary-text or raw-key entry point. [`docs/agent-control.md`](agent-control.md) owns the verb contract, the capability matrix, the relaunch transaction, and the fail-closed boundaries. ## Busy state is semantic, per adapter diff --git a/docs/configuration.md b/docs/configuration.md index 5513cfabd6f..c62bf92a6f5 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -83,6 +83,7 @@ Each effective `FM_HOME` contains private operational directories. `state/` holds runtime records: - Task metadata, append-only status events, and endpoint signals. +- Exact deliberately stood-down worker records under `state/.worker-state` (`bin/fm-worker-state-lib.sh`). - Watcher and wake-queue coordination, away-mode state, and generated Relay artifacts. - Inactive terminal-outcome receipts under `state/terminal-outcomes/`. - Enabled extension working namespaces under `state/extensions/`. @@ -2316,7 +2317,7 @@ FM_CODEX_WATCH_CHECKPOINT=180 # seconds per foreground watcher checkpoint in C FM_CODEX_WATCH_CHECKPOINT_AWAY=3600 # requested away checkpoint bound on a home that runs the supervision host; longer of this and attended bound, capped at 27000 FM_CREW_STATE_NM_TIMEOUT=10 # seconds allowed per no-mistakes query inside fm-crew-state.sh, and per state-database run-inventory read behind a capped AXI overview FM_TEARDOWN_NM_TIMEOUT=10 # seconds allowed per no-mistakes query or abort inside fm-teardown.sh -FM_CREW_STATE_RUNS_LIMIT=200 # plain runs-ledger rows scanned for fallback attribution; does not change the CLI's AXI overview window (selection owner: bin/fm-nm-run-lib.sh) +FM_CREW_STATE_RUNS_LIMIT=200 # plain runs-ledger rows scanned for fallback attribution, and the corroboration window the shared branch-run verdict reads for `fm-control stand-down`, where a full window simply adds no run; does not change the CLI's AXI overview window (selection owner: bin/fm-nm-run-lib.sh) FM_TEARDOWN_NM_RUNS_LIMIT=200 # recent no-mistakes run rows scanned to prove an unresolved-head parked run belongs to teardown's task FM_CREW_STATE_BIN=bin/fm-crew-state.sh # test override for the current-state reader used by watcher triage: the working/paused classification, and the wedge timer's parked-gate wait evidence FM_MAIL_USER= # mail-plane IMAP/SMTP login, from .env or environment (docs/configuration.md "Mail plane") diff --git a/docs/scripts.md b/docs/scripts.md index 8d656ec203e..c68e9a58d82 100644 --- a/docs/scripts.md +++ b/docs/scripts.md @@ -124,7 +124,7 @@ The shared no-mistakes gate lifecycle boundary is summarized in [architecture.md | `fm-branch-outcome.sh` | Own the supervision branch's append-only outcome store, cursors, bounded status-coverage indexes, and session-start replay | | `fm-lease.sh` | Claim, release, inspect, and sweep per-task supervision leases | | `fm-lease-lib.sh` | One owner of the supervision lease contract and the main-only role-partition guards | -| `fm-control.sh` | Agent lifecycle control plane: allowlisted `interrupt`, `exit`, and transactional `relaunch` verbs for an exact task id ([agent-control.md](agent-control.md)) | +| `fm-control.sh` | Agent lifecycle control plane: allowlisted lifecycle and worker-state verbs for an exact task id ([agent-control.md](agent-control.md)) | | `fm-control-lib.sh` | One executable owner of the control-plane verb allowlist, per-harness interrupt/exit mechanics, per-backend capability, and the endpoint-absence proof both `exit` and `relaunch` read | | `fm-busy-lib.sh` | Single owner of the semantic busy-state contract: verdicts, source attribution, and per-harness sources | | `fm-busy-event.sh` | The only writer of a task's semantic busy-state record and native-harness progress marker; arms an incarnation and applies lifecycle events | diff --git a/tests/fm-contributions.test.sh b/tests/fm-contributions.test.sh index ce87baefafb..0199b3b2ba7 100755 --- a/tests/fm-contributions.test.sh +++ b/tests/fm-contributions.test.sh @@ -652,7 +652,7 @@ case "$fault:$*" in fail-late:'api repos/o/r/pulls/8/reviews?'*) clock_bump 100; printf 'HTTP 502\n' >&2; exit 1 ;; fail:'api repos/o/r/pulls/8/reviews?'*) printf 'HTTP 502\n' >&2; exit 1 ;; down:*) printf 'HTTP 502\n' >&2; exit 1 ;; - hang:'api repos/o/r/pulls/8') sleep 4 ;; + hang:'api repos/o/r/pulls/8/reviews?'*) sleep 30 ;; head:'pr view '*) printf '{"headRefOid":"%s","reviewDecision":"APPROVED"}\n' "$(printf 'b%.0s' $(seq 40))"; exit 0 ;; esac exec "$(dirname "$0")/gh-fixture" "$@" @@ -672,11 +672,20 @@ test_budget_exhaustion_keeps_prior_record() { # exhaust|hang wrap_forge "$home" mutate_record "$home" delivery '.records[0].checked_at="2026-09-15T08:00:00Z"' cp "$home/data/delivery/contributions.json" "$home/prior.json" - # Both modes freeze the clock: an unfrozen one can tick past a one-second - # budget before the first forge call, so nothing is ever observed. + # Both modes freeze the clock so the budget expires only where the fault + # decides, never from wall-clock time passing between forge calls. + # + # Neither mode may make its evidence depend on how fast a fixture process + # starts. Exhaustion is driven from inside the observation - `exhaust` jumps + # the frozen clock past the deadline, `hang` blocks a later call well past + # the bound - so the calls that prove the observation started are ordinary + # completed reads, not the read being killed. The budget is five seconds + # rather than one for the same reason: it is still `-le 5`, so a killed read + # is still classified as budget exhaustion rather than an unavailable forge, + # but a contended runner cannot spend the whole bound spawning the fixture. /bin/date +%s > "$home/forge/clock" printf '%s\n' "$mode" > "$home/forge/fault" - out=$(with_home "$home" env FM_CONTRIBUTIONS_BUDGET=1 "$ROOT/bin/fm-contributions.sh" poll) \ + out=$(with_home "$home" env FM_CONTRIBUTIONS_BUDGET=5 "$ROOT/bin/fm-contributions.sh" poll) \ || fail "poll failed when its budget ran out ($mode)" [ -z "$out" ] || fail "budget exhaustion ($mode) printed a wake line: $out" grep -F 'api repos/o/r/pulls/8' "$home/forge/calls" >/dev/null \ diff --git a/tests/fm-control-relaunch.test.sh b/tests/fm-control-relaunch.test.sh index 1242dc26bfe..abe68eb770b 100755 --- a/tests/fm-control-relaunch.test.sh +++ b/tests/fm-control-relaunch.test.sh @@ -43,9 +43,9 @@ TASK_TMPS=() relaunch_cleanup() { local d for d in "${TASK_TMPS[@]:-}"; do - [ -n "$d" ] && rm -rf "$d" + [ -n "$d" ] && fm_test_remove_tree "$d" done - rm -rf "$TMP_ROOT" + fm_test_remove_tree "$TMP_ROOT" } trap relaunch_cleanup EXIT @@ -193,6 +193,13 @@ new_case() { printf '%s\n' "fm-$id" > "$dir/fake/windows" printf '%s' fmses > "$dir/fake/session-name" make_tmux_stub "$dir" + # A no-op `no-mistakes`, so a stand-down here can never reach a real + # installation on the developer's PATH to ask about an in-flight run. + cat > "$dir/fakebin/no-mistakes" <<'SH' +#!/usr/bin/env bash +exit 0 +SH + chmod +x "$dir/fakebin/no-mistakes" printf '%s\n' "$dir" } @@ -390,6 +397,31 @@ test_same_harness_relaunch_keeps_identity_and_reuses_the_endpoint() { pass "fm-control relaunch: a same-harness relaunch replaces the agent in the same endpoint and worktree" } +test_relaunch_from_a_stood_down_worker_reuses_its_preserved_worktree() { + local dir out rc + dir=$(new_case stood-down-relaunch rl-parked) + add_ship_task "$dir" rl-parked claude + out=$(run_control "$dir" rl-parked stand-down); rc=$? + expect_code 0 "$rc" "stand-down should stop the worker before a later relaunch"$'\n'"$out" + [ -f "$dir/home/state/rl-parked.worker-state" ] \ + || fail "stand-down did not retain the durable worker-free declaration" + [ "$(cat "$dir/fake/command")" = zsh ] \ + || fail "stand-down did not leave the recorded endpoint agent-free" + out=$(run_control "$dir" rl-parked relaunch --note "resuming the held task"); rc=$? + expect_code 0 "$rc" "relaunch from a stood-down worker should succeed"$'\n'"$out" + [ ! -e "$dir/home/state/rl-parked.worker-state" ] \ + || fail "a replacement worker must clear the old no-worker declaration" + [ "$(meta_field "$dir" rl-parked worktree)" = "$dir/wt" ] \ + || fail "relaunch from stand-down must reuse the preserved worktree" + [ "$(meta_field "$dir" rl-parked window)" = fmses:fm-rl-parked ] \ + || fail "relaunch from stand-down must reuse the preserved endpoint" + [ "$(cat "$dir/fake/command")" = claude ] \ + || fail "relaunch from stand-down did not start the replacement worker" + assert_grep 'Firstmate operational input waiting: read' "$dir/fake/literal" \ + "the replacement must receive the persisted brief in the preserved worktree" + pass "fm-control relaunch: a stood-down task restarts in its preserved worktree and clears only the no-worker declaration" +} + test_relaunch_refuses_before_exit_when_the_composer_holds_pending_text() { local dir out rc dir=$(new_case pending-exit rl43) @@ -1401,6 +1433,73 @@ test_complete_journal_failure_rolls_back_from_durable_phase() { pass "fm-control relaunch: failed journal replacement preserves durable phase" } +# A relaunch that never publishes its replacement must hand the task back +# exactly as the operator left it. Without the restore, an aborted relaunch +# leaves the task both worker-free AND undeclared - the precise state that puts +# a held task back into false stale/wedge alarms. +test_prepublication_abort_restores_the_intentional_no_worker_record() { + local dir out rc real_mv meta record + dir=$(new_case standdownrollback rl31) + add_ship_task "$dir" rl31 claude + record="$dir/home/state/rl31.worker-state" + meta="$dir/home/state/rl31.meta" + out=$(run_control "$dir" rl31 stand-down); rc=$? + expect_code 0 "$rc" "the task should start from a published stand-down"$'\n'"$out" + [ -f "$record" ] || fail "stand-down did not publish its record" + real_mv=$(command -v mv) + make_mv_failure_stub "$dir" + out=$(FM_REAL_MV="$real_mv" FM_FAKE_META_PUBLISH_MV_FAIL="$meta" \ + run_control "$dir" rl31 relaunch --note "try to bring the held task back"); rc=$? + expect_code 1 "$rc" "a failed metadata publication should fail closed"$'\n'"$out" + [ -f "$record" ] \ + || fail "an aborted relaunch left the held task worker-free and undeclared" + assert_grep 'state=stood-down' "$record" \ + "the restored record must be the exact declaration the relaunch cleared" + assert_grep 'endpoint=fmses:fm-rl31' "$record" \ + "the restored record must stay bound to the preserved endpoint" + out=$(FM_REAL_MV="$real_mv" \ + run_control "$dir" rl31 relaunch --note "retry now that publication works"); rc=$? + expect_code 0 "$rc" "a retried relaunch should still work from the restored record"$'\n'"$out" + [ ! -e "$record" ] \ + || fail "a relaunch that really started a replacement must clear the declaration" + pass "fm-control relaunch: a pre-publication abort returns the task to its declared stand-down" +} + +# The refusal on an unprovable record is a REFUSAL, so it must leave the task +# exactly as it found it. Retiring the prior incarnation's wiring first would +# strand it: that exit path arms no replacement, so it rolls nothing back. +test_invalid_worker_state_refuses_before_retiring_prior_wiring() { + local dir out rc record wiring + dir=$(new_case standdowninvalid rl32) + add_ship_task "$dir" rl32 claude + record="$dir/home/state/rl32.worker-state" + wiring="$dir/wt/.claude/settings.local.json" + mkdir -p "$dir/wt/.claude" + printf '{"hooks":{}}\n' > "$wiring" + cat > "$record" <<'EOF' +schema=1 +task_id=rl32 +endpoint=fmses:fm-some-other-endpoint +state=stood-down +EOF + out=$(run_control "$dir" rl32 relaunch --note "relaunch behind an unprovable record"); rc=$? + expect_code 1 "$rc" "an unprovable worker-state record must refuse the relaunch"$'\n'"$out" + assert_contains "$out" "repair-worker-state" \ + "the refusal should name the supported reconciliation" + [ "$(cat "$dir/fake/command")" = claude ] \ + || fail "an invalid worker-state refusal must leave the original agent alive" + [ -f "$wiring" ] \ + || fail "a refusal that arms no replacement must not retire the prior incarnation's wiring" + [ "$(meta_field "$dir" rl32 harness)" = claude ] \ + || fail "a refused relaunch must leave the durable record untouched" + [ -f "$record" ] || fail "the refused relaunch must leave the record for repair" + out=$(run_control "$dir" rl32 repair-worker-state); rc=$? + expect_code 0 "$rc" "repair should reconcile the record"$'\n'"$out" + out=$(run_control "$dir" rl32 relaunch --note "retry once the record is reconciled"); rc=$? + expect_code 0 "$rc" "the relaunch should work once the record is reconciled"$'\n'"$out" + pass "fm-control relaunch: an unprovable worker-state record leaves the prior worker intact" +} + test_prepublication_abort_retires_replacement_wiring_and_busy_state() { local dir out rc real_mv meta dir=$(new_case prepublishcleanup rl28) @@ -2125,7 +2224,11 @@ test_herdr_relaunch_resumes_only_the_registered_pi_session() { } dir=$HERDR_CASE_DIR rm -f "$dir/fake/herdr-stopped" - sed -i 's/^harness=claude$/harness=pi/' "$dir/home/state/resume-$registered.meta" + printf '#!/usr/bin/env bash\nprintf "Options: --tui-mode\\n"\n' > "$dir/fakebin/pi" + chmod +x "$dir/fakebin/pi" + sed 's/^harness=claude$/harness=pi/' "$dir/home/state/resume-$registered.meta" \ + > "$dir/home/state/resume-$registered.meta.tmp" + mv "$dir/home/state/resume-$registered.meta.tmp" "$dir/home/state/resume-$registered.meta" # Keep the pane's status authority registered to an existing Pi session, # while process-info proves that its previous agent has exited. printf '{"result":{"agent":{"agent":"%s","agent_status":"idle","agent_session":{"kind":"path","value":"/tmp/pi-bound-session.jsonl"}}}}\n' \ @@ -2387,6 +2490,7 @@ test_relaunch_moves_a_drifted_item_back_in_flight() { } test_same_harness_relaunch_keeps_identity_and_reuses_the_endpoint +test_relaunch_from_a_stood_down_worker_reuses_its_preserved_worktree test_relaunch_refuses_before_exit_when_the_composer_holds_pending_text test_relaunch_refuses_before_exit_when_the_composer_state_is_unproven test_relaunch_from_linked_home_preserves_recorded_worktree @@ -2429,6 +2533,8 @@ test_post_publication_launch_failure_keeps_the_new_record test_stop_transport_failure_reconciles_a_dead_agent test_complete_journal_failure_rolls_back_from_durable_phase test_prepublication_abort_retires_replacement_wiring_and_busy_state +test_prepublication_abort_restores_the_intentional_no_worker_record +test_invalid_worker_state_refuses_before_retiring_prior_wiring test_journal_records_the_checkpoint_it_proved test_secondmate_relaunch_checkpoints_child_work_and_spares_the_charter test_secondmate_relaunch_refuses_an_unmarked_home diff --git a/tests/fm-control.test.sh b/tests/fm-control.test.sh index 832c3fd7a49..3ad5d2bdd93 100755 --- a/tests/fm-control.test.sh +++ b/tests/fm-control.test.sh @@ -31,6 +31,7 @@ SEND="$ROOT/bin/fm-send.sh" # fm_test_tmproot's own cleanup trap fires when its command substitution exits, # so recreate the root before resolving it and clean it up from this file's trap. TMP_ROOT=$(fm_test_tmproot fm-control) +fm_git_identity fmtest fmtest@example.invalid mkdir -p "$TMP_ROOT" TMP_ROOT=$(cd "$TMP_ROOT" && pwd) trap 'rm -rf "$TMP_ROOT"' EXIT @@ -169,7 +170,14 @@ case "${1:-}" in printf '1\n' fi exit 0 ;; - *pane_current_command*) cat "$D/command"; printf '\n'; exit 0 ;; + *pane_current_command*) + # The stand-down race: an agent that stops on its own after the + # transition record is published, between two agent-state reads. + if [ -n "${FM_FAKE_DIES_WHEN_STANDING_DOWN:-}" ] \ + && grep -qs 'state=standing-down' "$FM_HOME"/state/*.worker-state; then + printf 'zsh' > "$D/command" + fi + cat "$D/command"; printf '\n'; exit 0 ;; *pane_current_path*) cat "$D/cwd"; printf '\n'; exit 0 ;; esac done @@ -199,6 +207,39 @@ SH printf '%s\n' "$fb" } +# A fake `no-mistakes` answering both run-attribution surfaces +# bin/fm-nm-run-lib.sh queries: the TOON `axi status` for the current branch, +# and the plain-text top-level `runs` listing used when that answers about +# another branch. FM_FAKE_NM_FAIL makes both calls fail silently the way a +# timed-out CLI does, so a test can pose the unanswerable question; +# FM_FAKE_NM_UNREGISTERED reproduces the installed CLI's answer in a repository +# it holds no registration for - exit 1, no rows, and the not-initialized +# response on stdout. +add_axi_stub() { # + cat > "$1/fakebin/no-mistakes" <<'SH' +#!/usr/bin/env bash +set -u +[ -z "${FM_FAKE_NM_FAIL:-}" ] || exit 1 +if [ -n "${FM_FAKE_NM_UNREGISTERED:-}" ]; then + echo "error: repo not initialized (run 'no-mistakes init' first)" + echo "help[1]: Run \`no-mistakes init\` to set up the gate in this repository" + exit 1 +fi +if [ -n "${FM_FAKE_NM_ERR:-}" ]; then + echo "$FM_FAKE_NM_ERR" >&2 + exit 1 +fi +case "${1:-} ${2:-}" in + "axi status") printf '%s\n' "${FM_FAKE_AXI_STATUS:-}" ;; + "runs "*|"runs") + [ -z "${FM_FAKE_NM_FAIL_RUNS:-}" ] || exit 1 + printf '%s\n' "${FM_FAKE_RUNS_LIST:-}" ;; +esac +exit 0 +SH + chmod +x "$1/fakebin/no-mistakes" +} + # new_case -> echoes a case dir holding home/, fake/, and fakebin. new_case() { local dir="$TMP_ROOT/$1-$RANDOM" @@ -208,6 +249,9 @@ new_case() { printf 'zsh' > "$dir/fake/command" printf 'claude' > "$dir/fake/becomes" make_tmux_stub "$dir" >/dev/null + # Every case gets the env-driven run stub, so no test can reach a real + # no-mistakes installation on the developer's PATH. + add_axi_stub "$dir" printf '%s\n' "$dir" } @@ -574,12 +618,16 @@ test_unverified_state_backends_refuse_stop_verbs() { assert_contains "$out" "no recovery-grade agent-state classifier" \ "the $backend refusal should name the missing stop proof" [ -z "$(literals "$dir")" ] || fail "$backend must receive no exit command" + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "stand-down on $backend should refuse"$'\n'"$out" + assert_contains "$out" "worker-lifecycle control is not supported" \ + "the $backend stand-down refusal should name the unsupported lifecycle surface" out=$(run_control "$dir" t1 relaunch --note x); rc=$? expect_code 1 "$rc" "relaunch on $backend should refuse"$'\n'"$out" assert_contains "$out" "no recovery-grade agent-state classifier" \ "the $backend relaunch refusal should name the missing stop proof" done - pass "fm-control: a backend that cannot prove an agent stopped refuses exit and relaunch" + pass "fm-control: a backend that cannot prove an agent stopped refuses exit, stand-down, and relaunch" } test_state_verified_backends_are_exactly_tmux_and_herdr() { @@ -593,6 +641,711 @@ test_state_verified_backends_are_exactly_tmux_and_herdr() { pass "fm-control-lib: stop-proving verbs are gated on the backends that really classify agent state" } +test_worker_state_verbs_refuse_herdr() { + local dir out rc verb + for verb in stand-down repair-worker-state; do + dir=$(new_case "herdr-$verb") + add_task "$dir" t1 claude ship herdr "lab:pane-1" + { + echo "herdr_session=lab" + echo "herdr_workspace_id=workspace-1" + echo "herdr_tab_id=tab-1" + echo "herdr_pane_id=pane-1" + } >> "$dir/home/state/t1.meta" + out=$(run_control "$dir" t1 "$verb"); rc=$? + expect_code 1 "$rc" "$verb on herdr should refuse"$'\n'"$out" + assert_contains "$out" "worker-lifecycle control is not supported" \ + "$verb should name the unsupported Herdr lifecycle surface" + [ -z "$(literals "$dir")" ] || fail "$verb on herdr must send no lifecycle command" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "$verb on herdr must not publish or alter worker state" + done + pass "fm-control: worker-state verbs do not drive Herdr" +} + +test_herdr_relaunch_reaches_existing_validation_path() { + local dir out rc + dir=$(new_case herdr-relaunch) + add_task "$dir" t1 claude ship herdr "lab:pane-1" + { + echo "herdr_session=lab" + echo "herdr_workspace_id=workspace-1" + echo "herdr_tab_id=tab-1" + echo "herdr_pane_id=pane-1" + } >> "$dir/home/state/t1.meta" + out=$(run_control "$dir" t1 relaunch); rc=$? + expect_code 1 "$rc" "Herdr relaunch without a progress note should reach ordinary validation"$'\n'"$out" + assert_contains "$out" "relaunch of a ship task requires --note" \ + "Herdr relaunch should be admitted before relaunch-specific validation" + assert_not_contains "$out" "worker-lifecycle control is not supported" \ + "ordinary Herdr relaunch must not be rejected as worker-state control" + [ -z "$(literals "$dir")" ] || fail "a note refusal must send no lifecycle command" + pass "fm-control: ordinary Herdr relaunch remains admitted" +} + +test_stand_down_proves_stop_then_records_intent() { + local dir out rc record + dir=$(new_case stand-down) + add_task "$dir" t1 claude + alive_as "$dir" claude + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "stand-down should stop a live agent and record the hold"$'\n'"$out" + assert_contains "$out" "stood-down t1 harness=claude" \ + "stand-down should report the deliberate worker-free state" + [ "$(literals "$dir")" = /exit ] \ + || fail "stand-down must use the proven exit path before declaring the worker absent" + [ "$(cat "$dir/fake/command")" = zsh ] \ + || fail "stand-down must prove the agent stopped before publishing its record" + record="$dir/home/state/t1.worker-state" + [ -f "$record" ] || fail "stand-down did not write its durable worker-state record" + assert_grep 'schema=1' "$record" "worker-state record must carry its schema" + assert_grep 'task_id=t1' "$record" "worker-state record must bind the task id" + assert_grep 'endpoint=fmses:fm-t1' "$record" "worker-state record must bind the exact endpoint" + assert_grep 'state=stood-down' "$record" "worker-state record must be published only after the stop" + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "a proven stood-down task should be idempotent"$'\n'"$out" + assert_contains "$out" "already-stood-down t1" "repeat stand-down should not issue another exit" + [ "$(wc -l < "$dir/fake/literal" | tr -d ' ')" = 1 ] \ + || fail "an already stood-down task must not receive another exit command" + pass "fm-control stand-down: a proven exit becomes an exact durable no-worker declaration" +} + +test_stand_down_accepts_an_agent_that_stopped_during_its_own_exit() { + local dir out rc record + dir=$(new_case stand-down-race) + add_task "$dir" t1 claude + alive_as "$dir" claude + out=$(FM_FAKE_DIES_WHEN_STANDING_DOWN=1 run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "an agent that stops itself mid stand-down still meets the postcondition"$'\n'"$out" + assert_contains "$out" "stood-down t1 harness=claude" \ + "an idempotent already-stopped exit must not refuse the hold" + record="$dir/home/state/t1.worker-state" + assert_grep 'state=stood-down' "$record" \ + "the declaration must not be left stuck at the transitional record" + pass "fm-control stand-down: a self-stopping agent is a successful stop, not a refusal" +} + +test_stand_down_refuses_to_relabel_an_unexpected_dead_agent() { + local dir out rc + dir=$(new_case stand-down-dead) + add_task "$dir" t1 claude + alive_as "$dir" zsh + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "stand-down must not relabel an already-dead agent" + assert_contains "$out" "already dead without a stand-down record" \ + "the refusal should preserve the distinction between a deliberate stop and a possible worker failure" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "an unexpected dead agent must not gain an intentional stand-down record" + pass "fm-control stand-down: an unexplained dead agent remains eligible for recovery" +} + + +# A stand-down declaration is only honest while nothing more current contradicts +# it, so these pin the two ways the control plane refuses to invent one: an +# in-flight validation run still needs a worker at its gates, and an agent that +# is merely gone is not an agent that was deliberately dismissed. + +axi_run_toon() { # + cat <&1); rc=$? + expect_code 0 "$rc" "a home with no run CLI owns no run to protect"$'\n'"$out" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "a task in a home without the run CLI must still be able to declare a hold" + # Counterfactual: an installed CLI that cannot answer still refuses. + rm -f "$dir/home/state/t1.worker-state" + alive_as "$dir" claude + add_axi_stub "$dir" + out=$(FM_FAKE_NM_FAIL=1 run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "an installed CLI that cannot answer must still refuse"$'\n'"$out" + assert_contains "$out" "no-mistakes axi status" \ + "the refusal should name the branch read that went unanswered" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "an unanswered run check must publish no worker-state record" + [ "$(cat "$dir/fake/command")" = claude ] || fail "a refused hold must leave the worker alive" + pass "fm-control stand-down: an absent run CLI licenses the hold, an unanswered one does not" +} + +# firstmate supports project modes that never register with no-mistakes +# (direct-PR, local-only), and a home may hold a repo that was simply never +# initialised. The CLI answers those with a not-initialized error and no rows, +# and a repository that owns no registration owns no run - so that answer is a +# proof of quiet, not an unanswered question the operator can never clear. +test_stand_down_allows_a_project_with_no_run_registration() { + local dir out rc + dir=$(new_case stand-down-unregistered) + add_task "$dir" t1 claude + alive_as "$dir" claude + out=$(FM_FAKE_NM_FAIL=1 run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "a silent CLI failure must still refuse the hold"$'\n'"$out" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "an unanswered run check must publish no worker-state record" + out=$(FM_FAKE_NM_ERR="database not initialized: cannot open run store" \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "an unrelated CLI failure must not read as proof of no registration"$'\n'"$out" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "an unrelated CLI failure must publish no worker-state record" + out=$(TMPDIR="$dir/absent-tmp" FM_FAKE_NM_UNREGISTERED=1 run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "a check that cannot capture the CLI's own error must refuse rather than degrade"$'\n'"$out" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "an uncapturable run check must publish no worker-state record" + out=$(FM_FAKE_NM_ERR="repo not initialized (run 'no-mistakes init' first)" \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "the earlier stderr form of the unregistered response must still license the hold"$'\n'"$out" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "an unregistered response on stderr must remain a proof that the project owns no run" + rm -f "$dir/home/state/t1.worker-state" + alive_as "$dir" claude + out=$(FM_FAKE_NM_UNREGISTERED=1 run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "a repository with no run registration owns no run to protect"$'\n'"$out" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "an unregistered project must still be able to declare a hold" + pass "fm-control stand-down: a project with no run registration is held, while a silent failure still refuses" +} + +# The corroboration listing can only ADD a run. A row it cannot parse is a row +# that names nothing, so it neither attributes a run nor overturns a branch read +# that already answered - while a readable live row in the same listing still +# refuses the hold. +test_stand_down_reads_a_garbled_listing_row_as_no_run_at_all() { + local dir out rc head + dir=$(new_case stand-down-unreadable-inventory) + add_task "$dir" t1 claude + alive_as "$dir" claude + head=$(git -C "$dir/wt-t1" rev-parse HEAD) + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-t1" "$head" completed)" \ + FM_FAKE_RUNS_LIST="running +running task-t1 $(git -C "$dir/wt-t1" rev-parse --short HEAD) 2026-08-28 " \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "a readable live row beside a garbled one must still refuse the hold"$'\n'"$out" + assert_contains "$out" "active no-mistakes run" \ + "the refusal should name the run the listing did read" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "a live run must publish no worker-state record" + [ -z "$(literals "$dir")" ] || fail "a live run must not lose its worker to a hold" + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-t1" "$head" completed)" \ + FM_FAKE_RUNS_LIST="running" run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "a garbled row alone must not overturn a branch read that answered"$'\n'"$out" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "an unparsable corroboration row leaves the branch read's own answer standing" + pass "fm-control stand-down: a garbled listing row adds no run and overturns no answer" +} + +# The head rule exists to reject a HISTORICAL run on a reused branch. A run +# that is still going owns the branch however far local work has advanced past +# the commit it started on, so an unplaceable live run is doubt, not proof of +# safety - from either attribution source. +test_stand_down_refuses_a_live_run_it_cannot_place() { + local dir out rc base head + dir=$(new_case stand-down-unplaceable-run) + add_task "$dir" t1 claude + alive_as "$dir" claude + base=$(git -C "$dir/wt-t1" rev-parse --short HEAD) + git -C "$dir/wt-t1" commit -q --allow-empty -m "work on top of the running run" + head=$(git -C "$dir/wt-t1" rev-parse HEAD) + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-other" "$head" running)" \ + FM_FAKE_RUNS_LIST="running task-t1 $base 2026-08-28 " \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "a running row this worktree cannot place must refuse the hold"$'\n'"$out" + assert_contains "$out" "active no-mistakes run" \ + "a run in flight on this branch refuses however its head places" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "an unplaceable live run must publish no worker-state record" + [ -z "$(literals "$dir")" ] || fail "an unplaceable live run must not lose its worker" + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-t1" "$base" running)" \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "the same doubt from axi status must refuse too"$'\n'"$out" + assert_contains "$out" "cannot be placed" \ + "the axi-status refusal should also name the unplaceable run" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "an unplaceable live run must publish no worker-state record" + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-other" "$head" running)" \ + FM_FAKE_RUNS_LIST="completed task-t1 $base 2026-08-28 " \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "an unplaceable TERMINAL row is history, and history does not block a hold"$'\n'"$out" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "a historical run on a reused branch still leaves the task free to be stood down" + pass "fm-control stand-down: a live run that cannot be placed is doubt, not proof of safety" +} + +# The mirror of the rule above. An unplaceable TERMINAL row is history that +# happens to live elsewhere - a pipeline lane head that never reached this +# worktree is routine - and history answers the only question a hold depends +# on, so it must not refuse a hold that has no run to finish or abort. +test_stand_down_allows_a_terminal_run_whose_head_never_reached_here() { + local dir out rc head + dir=$(new_case stand-down-terminal-lane-head) + add_task "$dir" t1 claude + alive_as "$dir" claude + head=$(git -C "$dir/wt-t1" rev-parse HEAD) + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-other" "$head" running)" \ + FM_FAKE_RUNS_LIST="completed task-t1 deadbeef1 2026-08-28 " \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "a finished run whose head is not a local object must not block the hold"$'\n'"$out" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "the hold publishes once the newest run for the branch is proven finished" + pass "fm-control stand-down: a terminal run whose head never reached this worktree is history, not doubt" +} + +# The listing is ordered newest first, so only the branch's NEWEST row answers +# whether a run is in flight. A run killed before it reached a terminal status +# leaves its row saying `running` forever; the later run that superseded it is +# the current truth, and an older live row must not displace it - the same rule +# the worktree parser and the overview selector already follow. +test_stand_down_reads_only_the_newest_ledger_row_for_the_branch() { + local dir out rc head short + dir=$(new_case stand-down-newest-ledger-row) + add_task "$dir" t1 claude + alive_as "$dir" claude + head=$(git -C "$dir/wt-t1" rev-parse HEAD) + short=$(git -C "$dir/wt-t1" rev-parse --short HEAD) + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-other" "$head" running)" \ + FM_FAKE_RUNS_LIST="completed task-t1 $short 2026-08-28 +running task-t1 $short 2026-08-27 " \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "a stale live row beneath the branch's newest finished run must not block the hold"$'\n'"$out" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "the branch's newest run being finished leaves the task free to be stood down" + rm -f "$dir/home/state/t1.worker-state" + : > "$dir/fake/literal" + alive_as "$dir" claude + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-other" "$head" running)" \ + FM_FAKE_RUNS_LIST="running task-t1 $short 2026-08-28 +completed task-t1 $short 2026-08-27 " \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "a newest live row must still refuse the hold"$'\n'"$out" + assert_contains "$out" "active no-mistakes run" \ + "the refusal should name the run that still needs a worker" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "a live run must publish no worker-state record" + [ -z "$(literals "$dir")" ] || fail "a live run must not lose its worker to a hold" + pass "fm-control stand-down: corroboration reads the branch's newest run, not its history" +} + +# `axi status` reports the most recent run, so a finished answer for this +# branch is no more proof than a finished listing row: an earlier run can still +# be in flight behind it. Only the whole-branch listing can prove the negative, +# so both sources have to agree before a hold is licensed. +test_stand_down_refuses_a_live_run_behind_a_terminal_axi_answer() { + local dir out rc head short + dir=$(new_case stand-down-live-behind-terminal-axi) + add_task "$dir" t1 claude + alive_as "$dir" claude + head=$(git -C "$dir/wt-t1" rev-parse HEAD) + short=$(git -C "$dir/wt-t1" rev-parse --short HEAD) + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-t1" "$head" completed)" \ + FM_FAKE_RUNS_LIST="running task-t1 $short 2026-08-27 " \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "a finished axi answer must not settle the branch while a run is live"$'\n'"$out" + assert_contains "$out" "active no-mistakes run" \ + "the refusal should name the run that still needs a worker" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "a live run behind a finished answer must publish no worker-state record" + [ -z "$(literals "$dir")" ] || fail "a live run must not lose its worker to a hold" + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-t1" "$head" completed)" \ + FM_FAKE_NM_FAIL_RUNS=1 run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "corroboration that could not be read must not overturn the branch read"$'\n'"$out" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "a readable branch read with no run in flight licenses the hold on its own" + pass "fm-control stand-down: the listing adds a live run, and its silence never overturns the branch read" +} + +# `no-mistakes runs` offers only --limit: no pagination, no end-of-list marker, +# and the window is exactly full on any repository whose history has reached it - +# the eventual steady state of every repository. Corroboration that came back +# full adds nothing, so it can neither prove nor unprove the branch, and the +# branch read's own answer stands; a live row INSIDE the window still refuses. +test_stand_down_takes_a_full_run_window_as_no_added_run() { + local dir out rc head + dir=$(new_case stand-down-full-window) + add_task "$dir" t1 claude + alive_as "$dir" claude + head=$(git -C "$dir/wt-t1" rev-parse HEAD) + out=$(FM_CREW_STATE_RUNS_LIMIT=3 \ + FM_FAKE_AXI_STATUS="$(axi_run_toon "task-other" "$head" running)" \ + FM_FAKE_RUNS_LIST="completed task-other aaaaaaa1 2026-08-28 +completed task-other aaaaaaa2 2026-08-28 +completed task-other aaaaaaa3 2026-08-28" \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "a full window must not refuse a branch read that already answered"$'\n'"$out" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "a mature repository's always-full window must not make the hold impossible" + # Counterfactual: a live row for this branch inside the same full window still + # refuses, so the window is read, not ignored. + rm -f "$dir/home/state/t1.worker-state" + : > "$dir/fake/literal" + alive_as "$dir" claude + out=$(FM_CREW_STATE_RUNS_LIMIT=3 \ + FM_FAKE_AXI_STATUS="$(axi_run_toon "task-other" "$head" running)" \ + FM_FAKE_RUNS_LIST="completed task-other aaaaaaa1 2026-08-28 +running task-t1 $(git -C "$dir/wt-t1" rev-parse --short HEAD) 2026-08-28 +completed task-other aaaaaaa3 2026-08-28" \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "a live row inside a full window must still refuse the hold"$'\n'"$out" + assert_contains "$out" "active no-mistakes run" \ + "the refusal should name the run the full window did show" + [ -z "$(literals "$dir")" ] || fail "a live run must not lose its worker to a hold" + pass "fm-control stand-down: a full window adds no run, and the live row it shows still refuses" +} + +# A live run the listing names is a run in flight whether or not this worktree +# can place its head; when the BRANCH READ is the source of that doubt, the +# refusal must name the run it could not place rather than a listing that was +# never the question. +test_stand_down_refusal_names_the_unplaceable_live_run_not_the_listing() { + local dir out rc head + dir=$(new_case stand-down-unplaceable-live-name) + add_task "$dir" t1 claude + alive_as "$dir" claude + head=$(git -C "$dir/wt-t1" rev-parse HEAD) + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-other" "$head" running)" \ + FM_FAKE_RUNS_LIST="running task-t1 deadbeef1 2026-08-28 " \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "a live run with an unplaceable head must refuse the hold"$'\n'"$out" + assert_contains "$out" "active no-mistakes run" \ + "a run in flight the listing named refuses however its head places" + assert_not_contains "$out" "could not answer" \ + "a listing that answered must not be reported as unable to answer" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "an unplaceable live run must publish no worker-state record" + git -C "$dir/wt-t1" commit -q --allow-empty -m "work on top of the running run" + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-t1" "$head" running)" \ + FM_FAKE_NM_FAIL_RUNS=1 run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "an unplaceable live run must refuse even when the listing cannot answer"$'\n'"$out" + assert_contains "$out" "cannot be placed" \ + "a named live run outranks the generic unanswered-listing reason" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "an unplaceable live run must publish no worker-state record" + pass "fm-control stand-down: an unplaceable live run is named, not misreported as an unanswered check" +} + +# The preserved local copy IS the hold. If it cannot be read, neither can the +# run state that decides whether the hold is safe to declare. +test_stand_down_refuses_when_the_worktree_cannot_be_read() { + local dir out rc + dir=$(new_case stand-down-no-worktree) + add_task "$dir" t1 claude + alive_as "$dir" claude + mv "$dir/wt-t1" "$dir/wt-t1-moved" + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "an absent worktree must refuse the hold"$'\n'"$out" + assert_contains "$out" "absent or unreadable" \ + "the refusal should name the worktree that could not be read" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "an unreadable worktree must publish no worker-state record" + [ -z "$(literals "$dir")" ] || fail "an unreadable worktree must not lose its worker" + mv "$dir/wt-t1-moved" "$dir/wt-t1" + # A ship sits at a detached HEAD from spawn until its worker's first + # `git checkout -b`, which is exactly when an early hold is most likely. + git -C "$dir/wt-t1" checkout -q --detach + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "a worktree with no branch owns no run, so the hold proceeds"$'\n'"$out" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "a just-spawned ship must still be declarable as deliberately worker-free" + pass "fm-control stand-down: an unreadable worktree refuses, while a branchless one owns no run" +} + +# A steer nobody has acknowledged is work the hold would silence: the watcher's +# re-ring ladder stops for a stood-down window and fm-send refuses to add to it. +test_stand_down_refuses_while_an_instruction_is_unacknowledged() { + local dir out rc rec + dir=$(new_case stand-down-pending-steer) + add_task "$dir" t1 claude + alive_as "$dir" claude + rec=$(bash -c '. "$1/bin/fm-task-inbox-lib.sh"; fm_task_inbox_write "$2" t1 "rebase onto main before you stop"' _ "$ROOT" "$dir/home/state") + [ -n "$rec" ] && [ -f "$rec" ] || fail "could not enqueue the durable instruction the test needs" + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "an unacknowledged instruction must refuse the hold"$'\n'"$out" + assert_contains "$out" "$rec" "the refusal should name the instruction still waiting" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "a pending instruction must publish no worker-state record" + [ -z "$(literals "$dir")" ] || fail "a pending instruction must not lose its worker" + mv "$rec" "$dir/home/state/t1.inbox/handled/" + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "an acknowledged instruction should free the hold"$'\n'"$out" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "the hold publishes once nothing is waiting unread" + pass "fm-control stand-down: an unacknowledged instruction is handled or withdrawn first" +} + +# The active-run refusal is not a ship privilege: a scout checked out on a +# branch can own a run exactly like a ship can. Its ordinary scratch +# configuration - a detached HEAD, where no branch exists for a run to own - +# still stands down normally. +test_stand_down_checks_a_scout_on_a_branch_and_spares_a_detached_scratch() { + local dir out rc head + dir=$(new_case stand-down-scout) + add_task "$dir" t1 claude scout + alive_as "$dir" claude + head=$(git -C "$dir/wt-t1" rev-parse HEAD) + out=$(FM_FAKE_AXI_STATUS="$(axi_run_toon "task-t1" "$head" awaiting_approval)" \ + run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "a scout on a branch with an in-flight run must refuse the hold"$'\n'"$out" + assert_contains "$out" "active no-mistakes run" \ + "the scout refusal should name the run that still needs a worker" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "a refused scout stand-down must publish no worker-state record" + [ -z "$(literals "$dir")" ] || fail "a refused scout stand-down must not stop the worker" + git -C "$dir/wt-t1" checkout -q --detach + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "a scout's detached scratch worktree owns no branch, so the hold proceeds"$'\n'"$out" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "the ordinary scout hold still publishes its declaration" + pass "fm-control stand-down: a scout is run-checked like a ship, and its scratch worktree still holds" +} + +test_a_prior_exit_becomes_intentional_only_after_a_declared_hold() { + local dir out rc + dir=$(new_case stand-down-after-exit) + add_task "$dir" t1 claude + alive_as "$dir" claude + out=$(run_control "$dir" t1 exit); rc=$? + expect_code 0 "$rc" "the ordinary exit verb should stop the agent"$'\n'"$out" + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "deadness alone must not become a declaration of intent" + assert_contains "$out" "Declare the hold first" \ + "the refusal should name the explicit operator action that makes the stop intentional" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "an undeclared dead agent must not gain an intentional record" + printf 'paused: holding this task until the upstream API lands\n' > "$dir/home/state/t1.status" + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "a declared hold should make the prior exit declarable"$'\n'"$out" + assert_contains "$out" "stood-down t1" "the declared hold should publish the no-worker state" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "the published record must be the proven stood-down state" + [ "$(literals "$dir")" = /exit ] \ + || fail "declaring an already-stopped agent must not send a second exit command" + printf 'working: back on it\n' >> "$dir/home/state/t1.status" + out=$(run_control "$dir" t1 repair-worker-state); rc=$? + expect_code 0 "$rc" "repair reads the endpoint, not the status log"$'\n'"$out" + assert_contains "$out" "intact" \ + "an ordinary status append must not retroactively revoke a published declaration" + assert_grep 'state=stood-down' "$dir/home/state/t1.worker-state" \ + "the published record survives until reality contradicts it" + # Reversibility is observable at the next declaration, not in the record: with + # the worker back and the hold withdrawn from the status log, the same dead + # agent is no longer declarable. + alive_as "$dir" claude + out=$(run_control "$dir" t1 repair-worker-state); rc=$? + expect_code 0 "$rc" "a live worker should reconcile the record"$'\n'"$out" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "a live worker must clear the declaration it contradicts" + alive_as "$dir" zsh + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "a withdrawn hold must stop licensing an intentional declaration"$'\n'"$out" + assert_contains "$out" "Declare the hold first" \ + "the withdrawn hold should send the operator back to the explicit declaration" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "a withdrawn hold must not publish a new intentional record" + pass "fm-control stand-down: an ordinary prior exit becomes intentional only through an explicit reversible hold" +} + +test_repair_clears_a_declaration_a_live_agent_contradicts() { + local dir out rc + dir=$(new_case repair-live) + add_task "$dir" t1 claude + alive_as "$dir" claude + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "stand-down should publish the record first"$'\n'"$out" + alive_as "$dir" claude + out=$(run_control "$dir" t1 repair-worker-state); rc=$? + expect_code 0 "$rc" "repair should reconcile a record reality contradicts"$'\n'"$out" + assert_contains "$out" "cleared-live-worker" "repair should report what it reconciled" + assert_contains "$out" "live agent" "repair should report the discrepancy to the operator" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "a declaration a live agent contradicts must not survive repair" + out=$(run_control "$dir" t1 repair-worker-state); rc=$? + expect_code 0 "$rc" "repair must be idempotent"$'\n'"$out" + assert_contains "$out" "no-record" "a second repair should be a no-op" + pass "fm-control repair-worker-state: reality wins over a stale declaration, idempotently" +} + +test_repair_clears_an_unprovable_record_without_inferring_intent() { + local dir out rc + dir=$(new_case repair-invalid) + add_task "$dir" t1 claude + alive_as "$dir" zsh + cat > "$dir/home/state/t1.worker-state" <<'EOF' +schema=1 +task_id=t1 +endpoint=fmses:fm-some-other-task +state=stood-down +EOF + out=$(run_control "$dir" t1 repair-worker-state); rc=$? + expect_code 0 "$rc" "repair should resolve a record bound to another endpoint"$'\n'"$out" + assert_contains "$out" "cleared-invalid" "repair should report the unprovable record it removed" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "an unprovable record must not survive repair" + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 1 "$rc" "repair must not leave a dead endpoint declared intentional" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "repair must never create a suppression from a dead endpoint alone" + pass "fm-control repair-worker-state: an unprovable record is cleared toward supervision, never toward intent" +} + +test_repair_clears_a_record_when_the_endpoint_vanished() { + local dir out rc + dir=$(new_case repair-missing) + add_task "$dir" t1 claude + alive_as "$dir" claude + out=$(run_control "$dir" t1 stand-down); rc=$? + expect_code 0 "$rc" "stand-down should publish the record first"$'\n'"$out" + out=$(run_control "$dir" t1 repair-worker-state); rc=$? + expect_code 0 "$rc" "repair should retain a declaration for a proven dead endpoint"$'\n'"$out" + assert_contains "$out" "intact agent-state=dead" \ + "repair should retain only the positively proved dead case" + [ -e "$dir/home/state/t1.worker-state" ] \ + || fail "a declaration for a proven dead endpoint should remain intact" + : > "$dir/fake/windows" + out=$(run_control "$dir" t1 repair-worker-state); rc=$? + expect_code 0 "$rc" "repair should clear a declaration whose endpoint vanished"$'\n'"$out" + assert_contains "$out" "cleared-unproven-endpoint" \ + "repair should report that endpoint absence was not proved dead" + assert_contains "$out" "agent-state=missing" \ + "repair should report the vanished endpoint observation" + [ ! -e "$dir/home/state/t1.worker-state" ] \ + || fail "a declaration for a vanished endpoint must not survive repair" + pass "fm-control repair-worker-state: a vanished endpoint returns to supervision" +} + # --- 3. exact-id scoping ---------------------------------------------------- test_window_label_is_refused_with_the_exact_id() { @@ -652,7 +1405,7 @@ test_record_bound_to_another_task_is_refused() { # refuses, and none of them reaches a local endpoint. test_remote_secondmate_is_refused_by_placement() { local dir out rc verb - for verb in interrupt exit relaunch; do + for verb in interrupt exit stand-down relaunch; do dir=$(new_case "remote-$verb") add_task "$dir" t1 claude secondmate alive_as "$dir" claude @@ -691,7 +1444,7 @@ hold_lifecycle_lock() { # test_interrupt_and_exit_lock_before_task_state_resolution() { local case_dir out rc verb lifecycle_lock_path holder i - for verb in interrupt exit; do + for verb in interrupt exit stand-down; do case_dir=$(new_case "locked-$verb") add_task "$case_dir" t1 claude alive_as "$case_dir" claude @@ -716,7 +1469,7 @@ test_interrupt_and_exit_lock_before_task_state_resolution() { [ -z "$(literals "$case_dir")" ] || fail "contended $verb must type no command" [ -z "$(keys_sent "$case_dir")" ] || fail "contended $verb must send no control key" done - pass "fm-control: interrupt and exit lock before task-state resolution" + pass "fm-control: interrupt, exit, and stand-down lock before task-state resolution" } # --- 4. verb allowlist ------------------------------------------------------ @@ -1085,6 +1838,32 @@ test_harness_kind_capability test_orca_refuses_an_escape_harness_interrupt test_unverified_state_backends_refuse_stop_verbs test_state_verified_backends_are_exactly_tmux_and_herdr +test_worker_state_verbs_refuse_herdr +test_herdr_relaunch_reaches_existing_validation_path +test_stand_down_proves_stop_then_records_intent +test_stand_down_accepts_an_agent_that_stopped_during_its_own_exit +test_stand_down_refuses_to_relabel_an_unexpected_dead_agent +test_stand_down_refuses_while_the_task_owns_an_active_run +test_stand_down_allows_a_terminal_run_for_the_same_task +test_stand_down_refuses_a_run_only_the_runs_list_can_attribute +test_stand_down_refuses_when_no_run_check_can_answer +test_stand_down_refuses_an_active_status_without_a_placeable_branch +test_stand_down_allows_a_home_without_the_run_cli +test_stand_down_allows_a_project_with_no_run_registration +test_stand_down_reads_a_garbled_listing_row_as_no_run_at_all +test_stand_down_refuses_a_live_run_it_cannot_place +test_stand_down_allows_a_terminal_run_whose_head_never_reached_here +test_stand_down_reads_only_the_newest_ledger_row_for_the_branch +test_stand_down_refuses_a_live_run_behind_a_terminal_axi_answer +test_stand_down_takes_a_full_run_window_as_no_added_run +test_stand_down_refusal_names_the_unplaceable_live_run_not_the_listing +test_stand_down_refuses_when_the_worktree_cannot_be_read +test_stand_down_refuses_while_an_instruction_is_unacknowledged +test_stand_down_checks_a_scout_on_a_branch_and_spares_a_detached_scratch +test_a_prior_exit_becomes_intentional_only_after_a_declared_hold +test_repair_clears_a_declaration_a_live_agent_contradicts +test_repair_clears_an_unprovable_record_without_inferring_intent +test_repair_clears_a_record_when_the_endpoint_vanished test_window_label_is_refused_with_the_exact_id test_explicit_endpoint_is_refused test_unknown_task_is_refused diff --git a/tests/fm-crew-state.test.sh b/tests/fm-crew-state.test.sh index bf7d7bc5306..724021142cb 100755 --- a/tests/fm-crew-state.test.sh +++ b/tests/fm-crew-state.test.sh @@ -209,13 +209,22 @@ set -u [ "${FM_FAKE_TMUX_UNREADABLE:-0}" = 1 ] && { printf 'no current client\n' >&2; exit 1; } case "${1:-}" in list-windows) - # A successful but empty inventory: it omits the crew's window, so absence - # is proved by the answer rather than by an addressed call failing. Only - # reached once display-message has already failed. - ;; + # The recovery-grade classifier reads the session inventory first. An + # inventory that omits the recorded window is `missing`, the default here, + # so absence is proved by the answer rather than by an addressed call + # failing. A test that needs the OTHER absent-agent verdict - the endpoint + # is still there and merely has no agent, `dead` - names its window in + # FM_FAKE_TMUX_WINDOWS. + printf '%s\n' "${FM_FAKE_TMUX_WINDOWS:-}" ;; display-message) [ "${FM_FAKE_TMUX_MISSING:-0}" = 1 ] && exit 1 - printf '%%1\n' ;; + fmt="" + for a in "$@"; do case "$a" in '#{'*) fmt=$a ;; esac; done + case "$fmt" in + '#{pane_current_command}') printf '%s\n' "${FM_FAKE_TMUX_CURRENT_COMMAND:-zsh}" ;; + '#{pane_tty}') printf '%s\n' "${FM_FAKE_TMUX_TTY:-}" ;; + *) printf '%%1\n' ;; + esac ;; capture-pane) [ "${FM_FAKE_TMUX_MISSING:-0}" = 1 ] && exit 1 if [ "${FM_FAKE_BUSY:-0}" = 1 ]; then printf 'work in progress\n%s\n' "${FM_FAKE_BUSY_TEXT:-esc to interrupt}" @@ -327,6 +336,9 @@ reset_fakes() { FM_FAKE_BUSY_TEXT= FM_FAKE_TMUX_MISSING=0 FM_FAKE_TMUX_UNREADABLE=0 + FM_FAKE_TMUX_WINDOWS="" + FM_FAKE_TMUX_CURRENT_COMMAND="" + FM_FAKE_TMUX_TTY="" FM_FAKE_HERDR_BUSY=0 FM_FAKE_HERDR_MISSING=0 FM_FAKE_HERDR_READ_FAIL=0 @@ -353,6 +365,7 @@ reset_fakes() { FM_FAKE_GERRIT_READ_LOG= unset FM_FAKE_PR_47_STATE FM_FAKE_PR_47_MERGED FM_FAKE_PR_48_STATE FM_FAKE_PR_48_MERGED export FM_FAKE_AXI_STATUS FM_FAKE_AXI_STATUS_RUN FM_FAKE_RUNS_LIST FM_FAKE_BUSY FM_FAKE_BUSY_TEXT FM_FAKE_TMUX_MISSING FM_FAKE_TMUX_UNREADABLE + export FM_FAKE_TMUX_WINDOWS FM_FAKE_TMUX_CURRENT_COMMAND FM_FAKE_TMUX_TTY export FM_FAKE_HERDR_BUSY FM_FAKE_HERDR_MISSING FM_FAKE_HERDR_READ_FAIL FM_FAKE_HERDR_HUSK FM_FAKE_HERDR_AGENT_STATUS FM_FAKE_HERDR_PROCESS FM_FAKE_HERDR_SHELL_PID FM_FAKE_CI_LOGS export FM_FAKE_DAEMON_DOWN FM_FAKE_DAEMON_TIMEOUT FM_FAKE_DAEMON_PROBE_LOG FM_FAKE_AXI_HOME export FM_FAKE_AXI_HOME_ERROR FM_FAKE_AXI_STATUS_RUN_ERROR FM_FAKE_AXI_STATUS_ERROR @@ -1600,6 +1613,214 @@ test_terminal_passed_with_failed_gitlab_read_reports_unknown() { pass "terminal passed run handles failed GitLab read" } +# A proven intentional stand-down outranks a HISTORICAL terminal validation +# result: that run describes the last worker incarnation, the record describes +# the task's current deliberate absence of a worker. +test_stood_down_worker_outranks_a_historical_failed_run() { + reset_fakes + local d out + d=$(new_case stood-down-failed-run) + make_repo_on_branch "$d/wt" fm/feat-stood-down + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-stood-down.meta" "window=fm:fm-feat-stood-down" "worktree=$d/wt" "kind=ship" + cat > "$d/state/feat-stood-down.worker-state" <<'EOF' +schema=1 +task_id=feat-stood-down +endpoint=fm:fm-feat-stood-down +state=stood-down +EOF + printf 'paused: waiting for an upstream maintainer\n' > "$d/state/feat-stood-down.status" + FM_FAKE_AXI_STATUS="$(run_failed fm/feat-stood-down)" + # The declared hold: the endpoint is still there and merely has no agent, so + # the preserved worktree and work can be relaunched in place. + FM_FAKE_TMUX_WINDOWS="fm-feat-stood-down" + out=$(run_crew_state "$d" feat-stood-down) + assert_contains "$out" "state: parked" \ + "a deliberately worker-free task must not render as its prior failed run" + assert_contains "$out" "source: worker-state" \ + "the current intentional worker state must name its own authoritative source" + assert_contains "$out" "worker deliberately stood down" \ + "the output must distinguish a healthy hold from a failed worker" + assert_not_contains "$out" "state: failed" \ + "a historical failed run must not mask the current deliberate stand-down" + pass "a stood-down worker state outranks historical failed validation state" +} + +# The endpoint half of the same rule. A stood-down record is a healthy park +# only while the endpoint it names is still there: a VANISHED endpoint cannot +# be relaunched in place, so it must be reported as an unknown that names the +# lost endpoint. This is the counterfactual for treating absence as healthy - +# if `missing` ever reads as a park again, this test sees "parked" instead. +test_a_vanished_endpoint_is_never_a_healthy_stood_down_hold() { + reset_fakes + local d out + d=$(new_case stood-down-endpoint-gone) + make_repo_on_branch "$d/wt" fm/feat-gone + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-gone.meta" "window=fm:fm-feat-gone" "worktree=$d/wt" "kind=ship" + cat > "$d/state/feat-gone.worker-state" <<'EOF' +schema=1 +task_id=feat-gone +endpoint=fm:fm-feat-gone +state=stood-down +EOF + printf 'paused: waiting for an upstream maintainer\n' > "$d/state/feat-gone.status" + FM_FAKE_TMUX_MISSING=1 + out=$(run_crew_state "$d" feat-gone) + assert_contains "$out" "state: unknown" \ + "a hold whose endpoint has vanished cannot be reported as healthy" + assert_contains "$out" "fm:fm-feat-gone" \ + "the report must name the endpoint that can no longer be relaunched" + assert_not_contains "$out" "state: parked" \ + "a vanished endpoint must not be reported as a deliberate park" + pass "a stood-down record whose endpoint vanished is reported as unknown, not as a healthy hold" +} + +# The base case the rule above protects: with no deliberate declaration at all, +# an absent worker is still a problem the reader must see. +test_an_absent_worker_without_a_declaration_is_still_reported() { + reset_fakes + local d out + d=$(new_case absent-undeclared) + make_repo_on_branch "$d/wt" fm/feat-absent + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-absent.meta" "window=fm:fm-feat-absent" "worktree=$d/wt" "kind=ship" + printf 'working: mid-task\n' > "$d/state/feat-absent.status" + FM_FAKE_TMUX_MISSING=1 + out=$(run_crew_state "$d" feat-absent) + assert_contains "$out" "state: unknown" \ + "an undeclared absent worker must be reported as unknown" + assert_contains "$out" "backend target gone: fm:fm-feat-absent" \ + "the report must name the endpoint that is gone" + assert_not_contains "$out" "worker deliberately stood down" \ + "an absent worker must never be described as a deliberate hold" + pass "an absent worker with no declaration is still reported as a problem" +} + +# The counterfactual for the rule above: the record outranks HISTORY, never an +# ACTIVE run. A run parked at a gate still owns the branch and still has work +# only a supervisor can action, so it must survive the record untouched. +test_active_run_outranks_a_stood_down_record() { + reset_fakes + local d out + d=$(new_case stood-down-active-run) + make_repo_on_branch "$d/wt" fm/feat-still-parked + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-still-parked.meta" "window=fm:fm-feat-still-parked" "worktree=$d/wt" "kind=ship" + cat > "$d/state/feat-still-parked.worker-state" <<'EOF' +schema=1 +task_id=feat-still-parked +endpoint=fm:fm-feat-still-parked +state=stood-down +EOF + FM_FAKE_AXI_STATUS="$(run_parked fm/feat-still-parked)" + FM_FAKE_TMUX_MISSING=1 + out=$(run_crew_state "$d" feat-still-parked) + assert_contains "$out" "source: run-step" \ + "an active run must keep reporting authority over a worker-state record" + assert_contains "$out" "2 finding(s)" \ + "the active run's own gate detail must survive the record" + assert_not_contains "$out" "worker deliberately stood down" \ + "a stand-down record must not replace an active run's own current state" + pass "an active run outranks a worker-state record" +} + +# A live run whose head this copy cannot resolve is still a live run on the +# task's preserved branch. The record describes the absence of a worker, never +# the absence of work, so the run must be reported with its details withheld +# instead of the task rendering as a healthy hold. +test_live_branch_run_outranks_a_stood_down_record_without_detail() { + reset_fakes + local d out + d=$(new_case stood-down-unplaceable-live-run) + make_repo_on_branch "$d/wt" fm/feat-unplaceable + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-unplaceable.meta" \ + "window=fm:fm-feat-unplaceable" "worktree=$d/wt" "kind=ship" + cat > "$d/state/feat-unplaceable.worker-state" <<'EOF' +schema=1 +task_id=feat-unplaceable +endpoint=fm:fm-feat-unplaceable +state=stood-down +EOF + printf 'paused: waiting for an upstream maintainer\n' > "$d/state/feat-unplaceable.status" + # The declared hold: the endpoint is still there with no agent in it. + FM_FAKE_TMUX_WINDOWS="fm-feat-unplaceable" + # The overview names another branch's run, and this branch's own live row + # carries a head object this copy never fetched, so ordinary attribution can + # place no run detail at all. + FM_FAKE_AXI_STATUS="$(run_running fm/other-crew)" + FM_FAKE_RUNS_LIST=" running fm/feat-unplaceable f0f0f0f0 2026-08-29 13:00" + out=$(run_crew_state "$d" feat-unplaceable) + assert_contains "$out" "state: working" \ + "a live run on the preserved branch must keep its own authority" + assert_contains "$out" "source: run-step" \ + "the live run, not the record, is the current statement about the task" + assert_contains "$out" "active run (details withheld)" \ + "an unplaceable live run must be reported without projecting details" + assert_not_contains "$out" "state: parked" \ + "a live run must never render as a healthy hold" + assert_not_contains "$out" "worker deliberately stood down" \ + "the stand-down record must not answer while a run is in flight" + pass "a live branch run outranks a stood-down record even without projectable detail" +} + +# The counterfactual for the arm above: a live run outranks a healthy HOLD, and +# nothing else. A vanished endpoint is not a hold - it is the one report that +# tells the operator the declared hold can no longer be resumed in place - so it +# keeps its own authority even while the branch owns a run. +test_a_live_run_does_not_hide_a_vanished_stood_down_endpoint() { + reset_fakes + local d out + d=$(new_case stood-down-gone-with-live-run) + make_repo_on_branch "$d/wt" fm/feat-gone-live + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-gone-live.meta" \ + "window=fm:fm-feat-gone-live" "worktree=$d/wt" "kind=ship" + cat > "$d/state/feat-gone-live.worker-state" <<'EOF' +schema=1 +task_id=feat-gone-live +endpoint=fm:fm-feat-gone-live +state=stood-down +EOF + printf 'paused: waiting for an upstream maintainer\n' > "$d/state/feat-gone-live.status" + FM_FAKE_TMUX_MISSING=1 + FM_FAKE_AXI_STATUS="$(run_running fm/other-crew)" + FM_FAKE_RUNS_LIST=" running fm/feat-gone-live f0f0f0f0 2026-08-29 13:00" + out=$(run_crew_state "$d" feat-gone-live) + assert_contains "$out" "state: unknown" \ + "a hold whose endpoint has vanished cannot be reported as healthy or as work" + assert_contains "$out" "fm:fm-feat-gone-live" \ + "the report must still name the endpoint that can no longer be relaunched" + assert_not_contains "$out" "state: working" \ + "a live run must not hide the lost endpoint" + pass "a live run does not suppress the vanished-endpoint report" +} + +# An unprovable record is a repair prompt, not a mask: it must never hide a +# real run outcome the reader can still act on. +test_invalid_worker_state_record_does_not_mask_a_failed_run() { + reset_fakes + local d out + d=$(new_case invalid-worker-state) + make_repo_on_branch "$d/wt" fm/feat-invalid-record + make_fakebin "$d" >/dev/null + fm_write_meta "$d/state/feat-invalid-record.meta" "window=fm:fm-feat-invalid-record" "worktree=$d/wt" "kind=ship" + cat > "$d/state/feat-invalid-record.worker-state" <<'EOF' +schema=1 +task_id=feat-invalid-record +endpoint=fm:fm-some-other-endpoint +state=stood-down +EOF + FM_FAKE_AXI_STATUS="$(run_failed fm/feat-invalid-record)" + FM_FAKE_TMUX_MISSING=1 + out=$(run_crew_state "$d" feat-invalid-record) + assert_contains "$out" "state: failed" \ + "a record that proves nothing must not mask a genuine failed run" + assert_contains "$out" "source: run-step" "the run remains the authoritative source" + pass "an unprovable worker-state record never masks a real run outcome" +} + test_terminal_passed_with_open_gerrit_change_does_not_claim_merged() { reset_fakes local d url read_log out @@ -5554,6 +5775,13 @@ test_terminal_passed_without_readable_pr_identity_reports_unknown test_terminal_passed_with_open_gitlab_mr_does_not_claim_merged test_terminal_passed_with_merged_gitlab_mr_reports_merged test_terminal_passed_with_failed_gitlab_read_reports_unknown +test_stood_down_worker_outranks_a_historical_failed_run +test_a_vanished_endpoint_is_never_a_healthy_stood_down_hold +test_an_absent_worker_without_a_declaration_is_still_reported +test_active_run_outranks_a_stood_down_record +test_live_branch_run_outranks_a_stood_down_record_without_detail +test_a_live_run_does_not_hide_a_vanished_stood_down_endpoint +test_invalid_worker_state_record_does_not_mask_a_failed_run test_terminal_passed_with_open_gerrit_change_does_not_claim_merged test_terminal_passed_with_merged_gerrit_change_reports_merged test_terminal_passed_with_unreadable_gerrit_change_reports_unknown diff --git a/tests/fm-send-inbox.test.sh b/tests/fm-send-inbox.test.sh index 2367771311d..6303c9ef223 100644 --- a/tests/fm-send-inbox.test.sh +++ b/tests/fm-send-inbox.test.sh @@ -68,7 +68,13 @@ case "${1:-}" in fi exit 0 ;; display-message) - for a in "$@"; do case "$a" in *cursor_y*) printf '1\n'; exit 0 ;; esac; done + for a in "$@"; do + case "$a" in + *cursor_y*) printf '1\n'; exit 0 ;; + *pane_tty*) [ "${FM_FAKE_TMUX_ENDPOINT:-}" = dead ] && { printf '\n'; exit 0; } ;; + *pane_current_command*) [ "${FM_FAKE_TMUX_ENDPOINT:-}" = dead ] && { printf 'zsh\n'; exit 0; } ;; + esac + done printf 'fakepane\n'; exit 0 ;; capture-pane) if [ "${FM_FAKE_TMUX_COMPOSER:-}" = pending ]; then @@ -77,7 +83,9 @@ case "${1:-}" in printf '╭────╮\n│ │\n╰────╯\n' fi exit 0 ;; - list-windows) printf 'fm-t1\n'; exit 0 ;; + list-windows) + [ "${FM_FAKE_TMUX_ENDPOINT:-}" = missing ] || printf 'fm-t1\n' + exit 0 ;; esac exit 0 SH @@ -339,6 +347,46 @@ test_key_path_never_touches_inbox() { pass "fm-send planes: the --key lifecycle path never touches the inbox" } +test_stood_down_task_refuses_an_unreceivable_steer() { + local dir err rc + dir=$(setup_case stood-down); err="$dir/send.err" + cat > "$dir/home/state/t1.worker-state" <<'EOF' +schema=1 +task_id=t1 +endpoint=sess:fm-t1 +state=stood-down +EOF + run_send "$dir" "$err" FM_FAKE_TMUX_ENDPOINT=dead -- t1 "do not leave this unread"; rc=$? + expect_code 1 "$rc" "a task with no worker should refuse a steer" + assert_contains "$(cat "$err")" "deliberately has no worker" \ + "the refusal should direct the caller to relaunch rather than silently queue an unread instruction" + [ ! -e "$dir/home/state/t1.inbox/001.msg" ] \ + || fail "a stood-down task must not receive an inbox record it cannot read" + [ ! -s "$dir/send.log" ] || fail "a stood-down task must not receive typed input" + pass "fm-send: a deliberately worker-free task refuses input until relaunch" +} + +test_stood_down_task_with_missing_endpoint_uses_durable_inbox() { + local dir err rc rec body + dir=$(setup_case stood-down-missing); err="$dir/send.err" + cat > "$dir/home/state/t1.worker-state" <<'EOF' +schema=1 +task_id=t1 +endpoint=sess:fm-t1 +state=stood-down +EOF + run_send "$dir" "$err" FM_FAKE_TMUX_ENDPOINT=missing -- t1 "recover this missing endpoint"; rc=$? + expect_code 0 "$rc" "a missing endpoint should stay on the ordinary supervision path" + rec="$dir/home/state/t1.inbox/001.msg" + [ -f "$rec" ] || fail "a missing endpoint did not receive its durable inbox record" + body=$(record_body _ "$rec") + [ "$body" = "recover this missing endpoint" ] \ + || fail "the missing endpoint's durable instruction differs: $body" + assert_not_contains "$(cat "$err")" "deliberately has no worker" \ + "a missing endpoint must not be treated as an intentional worker-free hold" + pass "fm-send: a missing stood-down endpoint stays under durable supervision" +} + test_secondmate_marker_and_enqueue_delivery() { local dir err body corr pr_rec delivered dir=$(setup_case secondmate) @@ -515,6 +563,8 @@ test_fire_and_forget_retry_stays_off_without_the_flag test_harness_invocations_stay_typed test_explicit_target_stays_typed test_key_path_never_touches_inbox +test_stood_down_task_refuses_an_unreceivable_steer +test_stood_down_task_with_missing_endpoint_uses_durable_inbox test_secondmate_marker_and_enqueue_delivery test_post_enqueue_bookkeeping_failure_is_not_retryable test_meta_lock_contention_fails_bounded diff --git a/tests/fm-task-inbox.test.sh b/tests/fm-task-inbox.test.sh index f447f1c344d..d18d9b849a1 100644 --- a/tests/fm-task-inbox.test.sh +++ b/tests/fm-task-inbox.test.sh @@ -698,6 +698,42 @@ test_watcher_rerings_idle_pane_quietly() { pass "watcher: an unhandled aged message on an idle pane re-rings without waking firstmate, and the ack silences it" } +# A deliberately worker-free task is exempt from pane-stale and wedge +# detection, and from nothing else. A steer that landed on the other side of +# the stand-down (the two guards that keep the inbox empty take different +# locks) must still be surfaced by the re-ring ladder rather than waiting +# silently for a relaunch. +test_watcher_ladder_still_runs_for_a_stood_down_task() { + local dir state out log pid rec i=0 + dir=$(setup_watch_case standdownladder) + state="$dir/state"; out="$dir/watch.out"; log="$dir/send.log"; : > "$log" + cat > "$state/t1.worker-state" <<'EOF' +schema=1 +task_id=t1 +endpoint=sess:fm-t1 +state=stood-down +EOF + rec=$(inbox_lib "$state" fm_task_inbox_write "$state" t1 "please continue") + age_path "$rec" + watch_bg "$state" "$dir/fakebin" "$out" \ + FM_SEND_LOG="$log" FM_FAKE_TMUX_CAPTURE="$(idle_capture "$dir")" \ + FM_FAKE_TMUX_AGENT=zsh \ + FM_TASK_INBOX_RING_MAX=1 + pid=$! + while [ "$i" -lt 150 ]; do + grep -qF "unread firstmate instruction: $rec" "$out" 2>/dev/null && break + kill -0 "$pid" 2>/dev/null || break + sleep 0.1 + i=$((i + 1)) + done + kill "$pid" 2>/dev/null; wait "$pid" 2>/dev/null + grep -qF "unread firstmate instruction: $rec" "$out" \ + || fail "a stood-down task must still surface its unread instruction:"$'\n'"$(cat "$out")" + grep -F "$rec" "$state/.wake-queue" >/dev/null \ + || fail "the unread instruction was not recorded in the durable queue" + pass "watcher: a stood-down task keeps its steering-inbox ladder, exempt only from pane-stale detection" +} + # A fresh process for each check proves the busy budget survives watcher restarts. busy_steer_check() { # [capture] [busy-max] PATH="$1/fakebin:$PATH" FM_STATE_OVERRIDE="$1/state" FM_SEND_LOG="$1/send.log" \ @@ -1197,6 +1233,7 @@ test_fire_and_forget_retry_is_owed_once test_fire_and_forget_retry_is_quiet_without_the_flag test_ring_ladder_policy test_watcher_rerings_idle_pane_quietly +test_watcher_ladder_still_runs_for_a_stood_down_task test_watcher_waits_on_busy_pane test_watcher_busy_budget_resets_on_ring_and_ack test_watcher_busy_bookkeeping_failure_surfaces diff --git a/tests/fm-watch-triage.test.sh b/tests/fm-watch-triage.test.sh index c2a2d60924d..aa131ef93ca 100755 --- a/tests/fm-watch-triage.test.sh +++ b/tests/fm-watch-triage.test.sh @@ -26,6 +26,7 @@ WATCH="$ROOT/bin/fm-watch.sh" DRAIN="$ROOT/bin/fm-wake-drain.sh" TMP_ROOT=$(fm_test_tmproot fm-watch-triage-tests) +TMP_ROOT=$(cd -P "$TMP_ROOT" && pwd -P) ack_stopped_cycle() { # local state=$1 err sequence generation @@ -2400,6 +2401,80 @@ test_nonterminal_stale_provably_working_absorbed_then_escalated() { pass "provably-working non-terminal stale is absorbed on first sight, then wedge-escalated past the threshold" } +# A deliberately stood-down worker has no agent by design. Its exact durable +# record suppresses pane hashing only after the recovery-grade classifier sees +# the shell left by fm-control exit, so a held task never starts a false wedge +# timer merely because its endpoint remains available for a later relaunch. +test_stood_down_dead_worker_is_excluded_from_stale_polling() { + local dir state fakebin out capture_file window pid + dir=$(make_case stood-down-dead); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-held" + printf 'static shell left for relaunch' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=claude\n' "$window" > "$state/held.meta" + cat > "$state/held.worker-state" < "$state/held.status" + printf '%s' "$(seen_sig "$state/held.status")" > "$state/.seen-held_status" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_FAKE_TMUX_CURRENT_COMMAND=zsh FM_STATE_OVERRIDE="$state" \ + FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid" + fail "a dead stood-down worker should keep the watcher running: $(cat "$out")" + fi + [ ! -s "$out" ] || { reap "$pid"; fail "a dead stood-down worker emitted a stale wake: $(cat "$out")"; } + [ ! -s "$state/.wake-queue" ] || { reap "$pid"; fail "a dead stood-down worker queued a stale wake"; } + [ ! -e "$state/.hash-test_fm-held" ] \ + || { reap "$pid"; fail "a dead stood-down worker was still pane-hashed for stale detection"; } + reap "$pid" + ack_stopped_cycle "$state" || fail "could not acknowledge the intentional held-task watcher stop" + pass "a proven stood-down worker is excluded from stale polling while its endpoint stays relaunchable" +} + +# The no-worker record is never a blanket exemption. This counterfactual puts a +# live agent behind an otherwise valid stood-down record and requires the real +# watcher process to surface the static worker as stale. If the new state ever +# suppresses a genuine live-worker wedge, this test blocks waiting for its wake. +test_stood_down_record_does_not_hide_a_live_worker_wedge() { + local dir state fakebin out capture_file window key pane_hash sig pid + dir=$(make_case stood-down-live); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-live-after-hold" + printf 'static live worker' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=claude\n' "$window" > "$state/live.meta" + cat > "$state/live.worker-state" < "$state/live.status" + sig=$(seen_sig "$state/live.status"); printf '%s' "$sig" > "$state/.seen-live_status" + key=$(printf '%s' "$window" | tr ':/. ' '____') + pane_hash=$(hash_text 'static live worker') + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_FAKE_CREW_STATE='state: unknown · source: none · no current-state source available' \ + FM_FAKE_TMUX_CURRENT_COMMAND=claude FM_STATE_OVERRIDE="$state" \ + FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 100 || fail "a live worker behind a stale stood-down record was not reported" + grep -Fx "stale: $window" "$out" >/dev/null \ + || fail "the live worker behind a stale stood-down record did not emit a stale wake: $(cat "$out")" + grep "$(printf '\tstale\t')" "$state/.wake-queue" | grep -F "$window" >/dev/null \ + || fail "the live worker wedge was not recorded in the durable queue" + pass "a stale stood-down record cannot hide a live worker wedge" +} + # --- non-terminal stale, crew NOT provably working: surfaced immediately ------ # The key requirement: a crew with no running pipeline that has gone quiet (and is # not busy) has stopped - it may be done via interactive menus, waiting, or wedged. @@ -6676,6 +6751,8 @@ test_permission_recovery_surfaces_preserved_status test_terminal_stale_surfaced test_stale_terminal_status_overridden_by_active_run test_nonterminal_stale_provably_working_absorbed_then_escalated +test_stood_down_dead_worker_is_excluded_from_stale_polling +test_stood_down_record_does_not_hide_a_live_worker_wedge test_wedge_escalation_marks_demand_deep_inspection_after_threshold test_wedge_escalation_resets_when_pane_becomes_active test_gone_endpoint_reports_once_instead_of_escalating_forever