From 88b2a9469418f27e078c3e138001c408c2ff1166 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Sun, 2 Aug 2026 15:52:47 -0700 Subject: [PATCH 01/35] fix(bin): correct session lock and attached watcher supervision (#1545) * fix(bin): identify harness sessions by path and report delivered wakes Two supervision faults, both reported by a contributor and both open on the default branch. Fault 1: the Stop auto-arm never claims the home. fm_harness_ancestry_pid() matched only the basename of `ps -o comm=`, and Claude Code's native installer names the per-session executable by its version (.../share/claude/versions/ 2.1.220), so that basename identifies nothing. Three real failure shapes follow: a version-named session is missed entirely and the hook exits 0 with the epoch never written (unconditional on Linux, where procps reports the kernel exec name and ignores argv[0]); a claude-named daemon that directly parents sessions wins the outermost-contiguous-claude rule ahead of the session itself; and a session that is both version-named and daemon-parented has its live lock reclaimed as stale and rewritten to the shared daemon pid, corrupting the home's ownership record. Harness identity now also reads whole components of the executable path and of argv[0], which is what both platforms still carry. Matching whole components only keeps that widening safe: bin/fm-claude-stop-autoarm.sh and ~/.claude/hooks scripts have no "claude" component. Ownership is then decided against the session's whole contiguous harness ancestry rather than one chosen pid, which is the honest form of the question the library already documents ("does the current process descend from that same harness?"). That subsumes the outermost-pid rule for Claude's nested bg-spare worker chain instead of reverting it, and lets a daemon-parented session recognize its own lock. Lock acquisition still writes the outermost pid of the run, the only pid that lives as long as the session. Fault 2: an attached arm reports a delivered cycle as FAILED. The watcher prints its one reason line to its own stdout, so only the arm that forked it can read that line; an arm that attached observes nothing but a released lock and called a completely successful cycle "cycle ended without an actionable reason". No supervision event was lost - the durable queue held it - but every harness protocol reads that line as "supervision is down" and directs a manual re-arm. The arm now resolves an unobservable close against the durable wake queue, which records every wake before the watcher prints it and whose sequence counter never rewinds, not even across a drain. A cycle the queue proves delivered a wake reports that wake and exits 0; a cycle whose records a handling turn already drained reports the delivery without inventing a reason line; only a cycle that delivered nothing is still the typed nonzero failure. Fixing it in the arm covers codex, opencode, pi, grok and kimi, not just the Claude Stop path. Regressions: tests/fm-session-lock-ancestry.test.sh pins both platforms' ps semantics behind a deterministic process table and runs the real Stop auto-arm in version-named, daemon-parented, and combined real process trees, each orphaned so the walk cannot escape the fixture. tests/fm-watch-arm.test.sh drives a real watcher and a real attached arm through a real wake. Every fault case fails on the previous code. * no-mistakes(review): Bind watcher delivery records to process identity * no-mistakes(review): Return validated watcher identity atomically * no-mistakes(review): Track watcher successors by PID and identity * no-mistakes(document): Consolidate watcher arm-cycle documentation ownership --- bin/fm-push-transition-lib.sh | 50 ++++ bin/fm-session-lock-lib.sh | 180 ++++++++---- bin/fm-test-run.sh | 3 +- bin/fm-wake-lib.sh | 12 +- bin/fm-watch-arm.sh | 83 ++++-- bin/fm-watch.sh | 4 +- docs/architecture.md | 2 +- docs/verification/supervision.md | 5 +- docs/watcher-continuity.md | 7 +- tests/fm-session-lock-ancestry.test.sh | 363 +++++++++++++++++++++++++ tests/fm-watch-arm.test.sh | 151 ++++++++++ 11 files changed, 779 insertions(+), 81 deletions(-) create mode 100755 tests/fm-session-lock-ancestry.test.sh create mode 100755 tests/fm-watch-arm.test.sh diff --git a/bin/fm-push-transition-lib.sh b/bin/fm-push-transition-lib.sh index 2738d7641de..f5711f21e7a 100644 --- a/bin/fm-push-transition-lib.sh +++ b/bin/fm-push-transition-lib.sh @@ -19,6 +19,54 @@ FM_PUSH_TRANSITION_LIB_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" TRIAGE_LOG="$STATE/.watch-triage.log" TRIAGE_LOG_MAX_BYTES=${FM_WATCH_TRIAGE_LOG_MAX_BYTES:-262144} FM_WAKE_POST_OUTPUT_ACTION= +FM_WATCH_DELIVERY_PID= +FM_WATCH_DELIVERY_IDENTITY= +WATCH_DELIVERY_LOG="$STATE/.watch-deliveries.log" +WATCH_DELIVERY_LOCK="$STATE/.watch-deliveries.lock" +WATCH_DELIVERY_MAX_BYTES=${FM_WATCH_DELIVERY_MAX_BYTES:-65536} +WATCH_DELIVERY_KEEP_LINES=${FM_WATCH_DELIVERY_KEEP_LINES:-64} +case "$WATCH_DELIVERY_MAX_BYTES" in ''|*[!0-9]*|0) WATCH_DELIVERY_MAX_BYTES=65536 ;; esac +case "$WATCH_DELIVERY_KEEP_LINES" in ''|*[!0-9]*|0) WATCH_DELIVERY_KEEP_LINES=64 ;; esac + +watch_delivery_clean_identity() { + printf '%s' "$1" | tr '\t\r\n' ' ' +} + +watch_delivery_clean_reason() { + printf '%s' "$1" | tr '\t\r\n' ' ' | cut -c1-4096 +} + +watch_delivery_publish() { + local reason=$1 i size tmp raw + [ -n "$FM_WATCH_DELIVERY_PID" ] || return 0 + [ -n "$FM_WATCH_DELIVERY_IDENTITY" ] || return 0 + i=0 + while ! fm_lock_try_acquire "$WATCH_DELIVERY_LOCK"; do + [ "$i" -lt 20 ] || return 0 + sleep 0.02 + i=$((i + 1)) + done + printf '%s\t%s\t%s\n' \ + "$FM_WATCH_DELIVERY_PID" \ + "$(watch_delivery_clean_identity "$FM_WATCH_DELIVERY_IDENTITY")" \ + "$(watch_delivery_clean_reason "$reason")" >> "$WATCH_DELIVERY_LOG" 2>/dev/null || true + size=$(wc -c < "$WATCH_DELIVERY_LOG" 2>/dev/null | tr -d '[:space:]') + case "$size" in + ''|*[!0-9]*) ;; + *) + if [ "$size" -ge "$WATCH_DELIVERY_MAX_BYTES" ]; then + tmp="$WATCH_DELIVERY_LOG.tmp.$FM_WATCH_DELIVERY_PID" + raw="$tmp.raw" + tail -n "$WATCH_DELIVERY_KEEP_LINES" "$WATCH_DELIVERY_LOG" 2>/dev/null \ + | tail -c "$WATCH_DELIVERY_MAX_BYTES" > "$raw" 2>/dev/null \ + && awk 'NR > 1 || /^[0-9]+\t/' "$raw" > "$tmp" 2>/dev/null \ + && mv -f "$tmp" "$WATCH_DELIVERY_LOG" 2>/dev/null + rm -f "$tmp" "$raw" 2>/dev/null || true + fi + ;; + esac + fm_lock_release "$WATCH_DELIVERY_LOCK" +} # Append one bounded best-effort line for an absorbed supervision event. triage_log() { @@ -39,9 +87,11 @@ wake() { heartbeat*) echo $(( $(cat "$STATE/.heartbeat-streak" 2>/dev/null || echo 0) + 1 )) > "$STATE/.heartbeat-streak" ;; *) echo 0 > "$STATE/.heartbeat-streak" ;; esac + trap '' HUP INT TERM [ -z "$FM_WAKE_POST_OUTPUT_ACTION" ] || trap '' PIPE if echo "$1"; then output_status=0 + watch_delivery_publish "$1" || true else output_status=1 fi diff --git a/bin/fm-session-lock-lib.sh b/bin/fm-session-lock-lib.sh index 8343a8efd97..0706b664c8d 100644 --- a/bin/fm-session-lock-lib.sh +++ b/bin/fm-session-lock-lib.sh @@ -11,57 +11,122 @@ # Known harness command names; extend when a new adapter is verified. FM_HARNESS_RE='claude|codex|opencode|grok|kimi|^pi$|^pi-signed$' -# Walk the current process ancestry (up to 16 hops) and print a harness pid. -# For every harness except Claude, the first match wins (innermost pid), which -# is where e.g. Pi's shared signed-wrapper ancestry actually holds the session: -# a "pi-signed" launcher can be the direct parent of the inner "pi" engine -# pid that owns the lock, and the wrapper pid above it is not that owner. -# Claude Code's bg-spare hook worker chain is the opposite shape: it nests -# several claude-named processes directly parent-child with no non-harness -# process between them, and the lock is held by the outermost pid of that -# run. So once a claude-named match is found, this keeps walking past it -# looking for a still-more-ancestral claude-named match, and stops the -# instant a non-match follows - never walking past that gap to an unrelated -# claude-named process further up the real process tree (e.g. the live -# session that launched a test as its own subprocess). The harness pid lives -# as long as the session, unlike the transient subshell pid of any one tool -# call. -fm_harness_ancestry_pid() { - local pid=$$ comm args best='' bc extending=0 hit=0 is_claude=0 +# The same harnesses as exact executable names. Keep in sync with +# FM_HARNESS_RE. Used only for the stricter path evidence below, where the +# loose regex would also match ordinary firstmate paths such as +# bin/fm-claude-stop-autoarm.sh. +FM_HARNESS_NAMES=(claude codex opencode grok kimi pi-signed pi) + +# Print the exact harness name carried by executable path $1 - its own basename +# or any directory component - or return 1. +# +# This exists because Claude Code's native installer names the per-session +# executable by its version (~/.local/share/claude/versions/2.1.220), so the +# basename identifies nothing while the install path still says claude. Matching +# whole path components only is what keeps that widening safe: an ordinary path +# such as bin/fm-claude-stop-autoarm.sh or ~/.claude/hooks/notify.sh has no +# "claude" component and is correctly not a harness process. +fm_harness_path_name() { # + local path=$1 name + [ -n "$path" ] || return 1 + for name in "${FM_HARNESS_NAMES[@]}"; do + case "/$path/" in + */"$name"/*) printf '%s' "$name"; return 0 ;; + esac + done + return 1 +} + +# True when the process described by command name $1 and full argument string $2 +# is a verified harness. Sets FM_HARNESS_IS_CLAUDE for the ancestry walk. +# +# Evidence, in order: +# 1. the basename of the reported command name, against FM_HARNESS_RE. +# 2. an exact harness component in that command path or in argv[0]. Both are +# needed because the two platforms report different things: macOS reports +# argv[0] in `ps -o comm=`, while procps on Linux reports the kernel exec +# name and ignores argv[0] entirely, so a version-named Claude Code binary +# is identified by its install path on macOS and by argv[0] on Linux. +# 3. a bare interpreter (node, python) running a harness script path. +FM_HARNESS_IS_CLAUDE=0 +fm_harness_process_matches() { # + local comm=$1 args=$2 base argv0 name + FM_HARNESS_IS_CLAUDE=0 + base=$(basename -- "$comm") + if printf '%s' "$base" | grep -qE "$FM_HARNESS_RE"; then + case "$base" in *claude*) FM_HARNESS_IS_CLAUDE=1 ;; esac + return 0 + fi + argv0=${args%% *} + if name=$(fm_harness_path_name "$comm") || name=$(fm_harness_path_name "$argv0"); then + case "$name" in claude) FM_HARNESS_IS_CLAUDE=1 ;; esac + return 0 + fi + # Bare interpreter (e.g. node): match the harness name in its script path. + case "$comm" in + *node*|*python*) + if printf '%s' "$args" | grep -qE "$FM_HARNESS_RE"; then + case "$args" in *claude*) FM_HARNESS_IS_CLAUDE=1 ;; esac + return 0 + fi + ;; + esac + return 1 +} + +# Walk the current process ancestry (up to 16 hops) and print this session's +# contiguous verified-harness ancestry, innermost pid first. +# +# The walk climbs freely until the first harness match, because the caller is +# normally an ordinary shell several levels below its session. After that first +# match it stops at the first non-harness ancestor, so it can never cross a gap +# into an unrelated harness further up the real process tree - for example the +# live session that launched a test as its own subprocess. +# +# For every harness except Claude the innermost match is the session, which is +# where e.g. Pi's shared signed-wrapper ancestry actually holds the lock: a +# "pi-signed" launcher can be the direct parent of the inner "pi" engine pid that +# owns the lock, and the wrapper pid above it is not that owner. Claude Code +# instead runs hooks several levels below the session inside its own nested +# worker chain (hook shell -> claude bg-spare -> claude bg-pty-host -> claude -> +# claude), with no non-harness process between them. Which pid in that run is the +# session cannot be read off the ancestry at all, so the whole contiguous run is +# reported and the callers below decide what they need from it. +fm_harness_ancestry_pids() { + local pid=$$ comm args extending=0 printed=0 for _ in 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16; do comm=$(ps -o comm= -p "$pid" 2>/dev/null) || break args=$(ps -o args= -p "$pid" 2>/dev/null) - bc=$(basename -- "$comm") - hit=0; is_claude=0 - if printf '%s' "$bc" | grep -qE "$FM_HARNESS_RE"; then - hit=1 - case "$bc" in *claude*) is_claude=1 ;; esac - else - # Bare interpreter (e.g. node): match the harness name in its script path. - case "$comm" in - *node*|*python*) - if printf '%s' "$args" | grep -qE "$FM_HARNESS_RE"; then - hit=1 - case "$args" in *claude*) is_claude=1 ;; esac - fi - ;; - esac - fi - if [ "$hit" -eq 1 ]; then - best="$pid" - if [ "$is_claude" -eq 1 ]; then - extending=1 - else - break - fi + if fm_harness_process_matches "$comm" "$args"; then + printf '%s\n' "$pid" + printed=1 + [ "$FM_HARNESS_IS_CLAUDE" -eq 1 ] || break + extending=1 elif [ "$extending" -eq 1 ]; then break fi pid=$(ps -o ppid= -p "$pid" 2>/dev/null | tr -d ' ') [ -n "$pid" ] && [ "$pid" -gt 1 ] || break done - [ -n "$best" ] && { echo "$best"; return 0; } - return 1 + [ "$printed" -eq 1 ] +} + +# Print the one pid that identifies this session when the session lock is being +# WRITTEN: the outermost pid of the contiguous run. That is the pid that lives as +# long as the session - a Claude worker several levels in is reaped when its hook +# returns, and a lock naming it would look stale moments later while the session +# is still running. Every non-Claude harness reports a single pid, so this is its +# innermost match unchanged. +fm_harness_ancestry_pid() { + local pids pid outermost='' + pids=$(fm_harness_ancestry_pids) || return 1 + while IFS= read -r pid; do + [ -n "$pid" ] && outermost=$pid + done </dev/null || return 1 comm=$(ps -o comm= -p "$pid" 2>/dev/null) || return 1 - if printf '%s' "$(basename -- "$comm")" | grep -qE "$FM_HARNESS_RE"; then - return 0 - fi - case "$comm" in - *node*|*python*) - args=$(ps -o args= -p "$pid" 2>/dev/null) - printf '%s' "$args" | grep -qE "$FM_HARNESS_RE" - ;; - *) return 1 ;; - esac + args=$(ps -o args= -p "$pid" 2>/dev/null) + fm_harness_process_matches "$comm" "$args" } -# True when state dir $1 holds a session lock whose pid is the harness ancestor +# True when state dir $1 holds a session lock whose pid is ANY harness ancestor # of the current process: this script runs inside the session that owns the -# home's fleet lock. A missing lock, a lock held by another live harness, or an +# home's fleet lock. Membership is the honest test of that question, because the +# lock owner sits at an unknown depth in a contiguous Claude run - it is the +# outermost pid when the hook fires inside the session's own nested worker chain, +# and an inner pid when a harness-named daemon parents the session. A missing +# lock, a malformed lock, a lock held by a harness outside this ancestry, or an # ancestry that cannot be resolved all fail closed. fm_session_lock_owned_by_self() { - local state=$1 lock_pid my_pid + local state=$1 lock_pid pids pid lock_pid=$(cat "$state/.lock" 2>/dev/null || true) case "$lock_pid" in ''|*[!0-9]*) return 1 ;; esac - my_pid=$(fm_harness_ancestry_pid) || return 1 - [ "$my_pid" = "$lock_pid" ] + pids=$(fm_harness_ancestry_pids) || return 1 + while IFS= read -r pid; do + [ "$pid" = "$lock_pid" ] && return 0 + done </dev/null || true) lock_path=$(cat "$lockdir/watcher-path" 2>/dev/null || true) @@ -88,22 +90,28 @@ fm_watcher_lock_matches_pid() { [ "$lock_path" = "$watch_path" ] || return 1 [ -n "$lock_identity" ] || return 1 current_identity=$(fm_pid_identity "$pid") || return 1 - [ "$current_identity" = "$lock_identity" ] + [ "$current_identity" = "$lock_identity" ] || return 1 + FM_WATCHER_MATCHED_IDENTITY=$lock_identity } FM_WATCHER_HEALTHY_PID= +FM_WATCHER_HEALTHY_IDENTITY= fm_watcher_healthy() { - local state=$1 watch_path=$2 grace=${3:-${FM_GUARD_GRACE:-300}} home=${4:-$FM_HOME} lockdir beat pid age + local state=$1 watch_path=$2 grace=${3:-${FM_GUARD_GRACE:-300}} home=${4:-$FM_HOME} lockdir beat pid identity age FM_WATCHER_HEALTHY_PID= + FM_WATCHER_HEALTHY_IDENTITY= lockdir="$state/.watch.lock" beat="$state/.last-watcher-beat" pid=$(cat "$lockdir/pid" 2>/dev/null || true) fm_pid_alive "$pid" || return 1 fm_watcher_lock_matches_pid "$state" "$watch_path" "$pid" "$home" || return 1 + identity=$FM_WATCHER_MATCHED_IDENTITY age=$(fm_path_age "$beat") [ "$age" -lt "$grace" ] || return 1 # shellcheck disable=SC2034 # Read by callers after fm_watcher_healthy returns. FM_WATCHER_HEALTHY_PID=$pid + # shellcheck disable=SC2034 # Read by callers after fm_watcher_healthy returns. + FM_WATCHER_HEALTHY_IDENTITY=$identity return 0 } diff --git a/bin/fm-watch-arm.sh b/bin/fm-watch-arm.sh index 3c2df49c891..81c09098d84 100755 --- a/bin/fm-watch-arm.sh +++ b/bin/fm-watch-arm.sh @@ -36,11 +36,13 @@ # stale-beacon or dead-pid holder either self-heals (the fresh child steals the # dead lock per the singleton self-eviction/steal path and is confirmed) or this # returns the FAILED line. On started it waits the child and propagates the wake -# reason; on attached it stays live across identity-matched successors. An -# attached cycle that ends without a healthy successor is a typed nonzero failure, -# never a clean empty completion. On FAILED it exits non-zero so the failure is -# loud. A live cycle already present means re-arm attaches - do not start a second -# watcher. +# reason; on attached it stays live across identity-matched successors. A cycle +# that ends with no reason line and no healthy successor is resolved against the +# watcher's identity-bound delivery record: a matching record reports that wake +# and exits 0, and only a cycle that delivered nothing is the typed nonzero +# failure. Neither is ever a clean empty completion. On FAILED it exits non-zero +# so the failure is loud. A live cycle already present means re-arm attaches - do +# not start a second watcher. # # Every observed watcher cycle appends one tab-separated lifecycle record to # state/.watch-cycle-exits.log. The arm layer owns that bounded ledger; it records @@ -99,8 +101,12 @@ lock_snapshot() { printf 'pid:%s|identity:%s' "$(cycle_clean_field "${pid:-none}")" "$(cycle_clean_field "${identity:-none}")" } +WATCH_DELIVERY_LOG="$STATE/.watch-deliveries.log" +WATCH_DELIVERY_LOCK="$STATE/.watch-deliveries.lock" + cycle_active=0 cycle_watcher_pid=none +cycle_watcher_identity=none cycle_origin=unknown cycle_started_at=0 cycle_lock_before='pid:none|identity:none' @@ -108,6 +114,7 @@ cycle_lock_before='pid:none|identity:none' cycle_begin() { cycle_watcher_pid=$1 cycle_origin=$2 + cycle_watcher_identity=$3 cycle_started_at=$(date +%s) cycle_lock_before=$(lock_snapshot) cycle_active=1 @@ -115,6 +122,9 @@ cycle_begin() { cycle_refresh_lock_before() { [ "$cycle_active" -eq 1 ] || return 0 + if [ "$HEALTHY_PID" = "$cycle_watcher_pid" ] && [ -n "$HEALTHY_IDENTITY" ]; then + cycle_watcher_identity=$HEALTHY_IDENTITY + fi cycle_lock_before=$(lock_snapshot) } @@ -229,10 +239,13 @@ clear_stale_recorded_watcher_lock() { # single honesty gate: a dead pid, a reused pid, or a stale beacon all fail it, so # this script can never report a watcher that is not really there. HEALTHY_PID= +HEALTHY_IDENTITY= healthy_watcher() { HEALTHY_PID= + HEALTHY_IDENTITY= fm_watcher_healthy "$STATE" "$WATCH" "$GRACE" "$FM_HOME" || return 1 HEALTHY_PID=$FM_WATCHER_HEALTHY_PID + HEALTHY_IDENTITY=$FM_WATCHER_HEALTHY_IDENTITY } report_attached() { @@ -261,17 +274,49 @@ fail_unexplained_cycle() { return 1 } +# Close a cycle whose reason line this arm could not read against the bounded +# terminal-delivery ledger the watcher publishes before releasing its lock. +close_unobserved_cycle() { + local i reason clean_identity record_pid record_identity record_reason + clean_identity=$(printf '%s' "$cycle_watcher_identity" | tr '\t\r\n' ' ') + i=0 + while ! fm_lock_try_acquire "$WATCH_DELIVERY_LOCK"; do + [ "$i" -lt 20 ] || { + fail_unexplained_cycle + return 1 + } + sleep 0.02 + i=$((i + 1)) + done + reason= + if [ -f "$WATCH_DELIVERY_LOG" ]; then + while IFS=$'\t' read -r record_pid record_identity record_reason; do + if [ "$record_pid" = "$cycle_watcher_pid" ] && [ "$record_identity" = "$clean_identity" ]; then + reason=$record_reason + fi + done < "$WATCH_DELIVERY_LOG" + fi + fm_lock_release "$WATCH_DELIVERY_LOCK" + if [ -n "$reason" ]; then + printf '%s\n' "$reason" + return 0 + fi + fail_unexplained_cycle + return 1 +} + # Stay alive across identity-matched healthy holders. If one cycle ends, attach -# to a verified successor. With no successor, fail loudly instead of returning a -# clean empty completion that an adapter could mistake for a no-op. +# to a verified successor. With no successor, report the wake that cycle durably +# delivered, or fail loudly - never a clean empty completion that an adapter could +# mistake for a no-op. attach_and_wait() { local attached_pid=$1 while :; do if healthy_watcher; then - if [ "$HEALTHY_PID" != "$attached_pid" ]; then + if [ "$HEALTHY_PID" != "$attached_pid" ] || [ "$HEALTHY_IDENTITY" != "$cycle_watcher_identity" ]; then cycle_log_append unknown unknown lock-replaced "attached:$HEALTHY_PID" attached_pid=$HEALTHY_PID - cycle_begin "$attached_pid" attached + cycle_begin "$attached_pid" attached "$HEALTHY_IDENTITY" report_attached fi sleep "$ATTACH_POLL" @@ -280,12 +325,15 @@ attach_and_wait() { if wait_for_healthy_successor; then cycle_log_append unknown unknown attached-cycle-ended "attached:$HEALTHY_PID" attached_pid=$HEALTHY_PID - cycle_begin "$attached_pid" attached + cycle_begin "$attached_pid" attached "$HEALTHY_IDENTITY" report_attached continue fi + if close_unobserved_cycle; then + cycle_log_append unknown unknown attached-delivered-wake none + return 0 + fi cycle_log_append unknown unknown attached-cycle-ended none - fail_unexplained_cycle return 1 done } @@ -357,7 +405,7 @@ fi # this home's watcher and wants a fresh one.) if [ "$mode" = arm ] && healthy_watcher; then cycle_mark_predecessor_successor "attached:$HEALTHY_PID" - cycle_begin "$HEALTHY_PID" attached + cycle_begin "$HEALTHY_PID" attached "$HEALTHY_IDENTITY" report_attached attach_and_wait "$HEALTHY_PID" exit $? @@ -401,7 +449,7 @@ child_out=$(mktemp "$STATE/.watch-arm-output.XXXXXX") || { } "$WATCH" >"$child_out" & child=$! -cycle_begin "$child" started +cycle_begin "$child" started "$(fm_pid_identity "$child" 2>/dev/null || true)" child_done=0 owned_child_finished() { @@ -426,16 +474,19 @@ owned_child_finished() { child_out= cycle_mark_predecessor_successor "attached:$HEALTHY_PID" report_attached - cycle_begin "$HEALTHY_PID" attached + cycle_begin "$HEALTHY_PID" attached "$HEALTHY_IDENTITY" attach_and_wait "$HEALTHY_PID" return $? fi - cycle_log_append "$rc" "$signal" unexpected-clean-exit none print_watch_output "$child_out" rm -f "$child_out" 2>/dev/null || true child= child_out= - fail_unexplained_cycle + if close_unobserved_cycle; then + cycle_log_append "$rc" "$signal" clean-exit-delivered-wake none + return 0 + fi + cycle_log_append "$rc" "$signal" unexpected-clean-exit none return 1 fi diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 2cbd3c77768..2f150af60d8 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -746,7 +746,9 @@ trap 'exit 1' HUP INT TERM WATCHER_PID=${BASHPID:-$$} printf '%s\n' "$FM_HOME" > "$WATCH_LOCK/fm-home" || true printf '%s\n' "$WATCH_PATH" > "$WATCH_LOCK/watcher-path" || true -fm_pid_identity "$WATCHER_PID" > "$WATCH_LOCK/pid-identity" 2>/dev/null || true +FM_WATCH_DELIVERY_PID=$WATCHER_PID +FM_WATCH_DELIVERY_IDENTITY=$(fm_pid_identity "$WATCHER_PID" 2>/dev/null || true) +printf '%s\n' "$FM_WATCH_DELIVERY_IDENTITY" > "$WATCH_LOCK/pid-identity" 2>/dev/null || true [ -e "$STATE/.last-heartbeat" ] || touch "$STATE/.last-heartbeat" diff --git a/docs/architecture.md b/docs/architecture.md index 95b0c2ff133..c94a95dd4f3 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -57,7 +57,7 @@ Optional X mode integrates with the watcher only after explicit opt-in; [configu At session start, `bin/fm-session-start.sh` emits exactly one primary-harness supervision block rendered by `bin/fm-supervision-instructions.sh` from `docs/supervision-protocols/`. That block owns the live wait shape for the running primary harness: Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi and pi-signed use the same two tracked primary extensions, and OpenCode uses its TUI plugin. `bin/fm-watch-arm.sh` remains the verified arm wrapper for protocols that call it; it forks the watcher as a tracked child, verifies it is genuinely alive with a fresh liveness beacon, and prints an honest `started`, `attached`, or nonzero `FAILED` status. -On `attached` it stays live across identity-matched successors, and an unexplained clean child close either attaches to a verified healthy successor or becomes the typed nonzero `watcher: FAILED - cycle ended without an actionable reason` result. +[`watcher-continuity.md`](watcher-continuity.md#arm-layer-cycle-contract) owns the arm layer's successor, terminal-delivery, and typed clean-close failure contract. The arm layer records one bounded lifecycle row per observed cycle in `state/.watch-cycle-exits.log`; `state/.watch-triage.log` remains exclusively the absorbed-wake debug log. Pi and OpenCode verify session-lock ownership and launch one singleton successor from their child-close handlers before delivering an actionable wake prompt, with bounded exponential retry for failed restoration. Claude's `bin/fm-claude-stop-autoarm.sh` hook fires on every Stop and, when the home is eligible and still needs supervision, claims one home-scoped cycle, foregrounds the arm wrapper, and translates an actionable close or typed failure into one exit-2 rewake. diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md index d2fdd5a09c9..00b13c1dd4a 100644 --- a/docs/verification/supervision.md +++ b/docs/verification/supervision.md @@ -127,7 +127,10 @@ The same run proved the Claude-compatible Stop entries stay inert under `GROK_AG The secondmate-home scope and manual-repair wake path were measured with Claude Code 2.1.207 on 2026-07-12, when a native background completion re-invoked the idle model with no human input. The current Stop-owned main/secondmate inclusion and child-worktree exclusion are covered deterministically by `tests/fm-claude-stop-autoarm.test.sh`. -On 2026-07-28 with Claude Code 2.1.205, `fm_harness_ancestry_pid()` in `bin/fm-session-lock-lib.sh` was fixed to resolve the outermost pid of a contiguous nested-harness run instead of the first match, so the Stop auto-arm correctly reaches the session's true lock owner through Claude Code's multi-level `bg-spare` hook worker chain. +Session-lock ownership in `bin/fm-session-lock-lib.sh` is decided against a session's whole contiguous harness ancestry rather than one chosen pid, so the Stop auto-arm reaches its lock owner wherever that owner sits: the outermost pid of Claude Code's multi-level `bg-spare` hook worker chain, or an inner pid when a harness-named daemon parents the session. +Harness identity is read from the executable path and `argv[0]` as well as the command basename, because Claude Code's native installer names the per-session executable by its version (`.../share/claude/versions/2.1.220`): `ps -o comm=` reports that path on macOS and the bare version string on Linux, and neither basename names a harness. +`tests/fm-session-lock-ancestry.test.sh` pins both platforms' reporting semantics behind a deterministic process table and runs the real Stop auto-arm in version-named, daemon-parented, and combined real process trees. +`tests/fm-watch-arm.test.sh` runs a real watcher and attached arm to verify that a delivered reason survives queue draining, while an unrelated queue append cannot make a watcher cycle that delivered nothing look successful. The Claude product live path ran with Claude Code 2.1.219 on 2026-07-24: diff --git a/docs/watcher-continuity.md b/docs/watcher-continuity.md index 52b3a9eec3a..952f7cca88d 100644 --- a/docs/watcher-continuity.md +++ b/docs/watcher-continuity.md @@ -39,8 +39,11 @@ The turn-end guard remains the final backstop rather than the normal continuity `bin/fm-watch-arm.sh` never returns a clean empty success. An actionable child output returns that reason normally. -A zero/empty child return rechecks the home lock and beacon, attaches to a verified healthy successor when one exists, or emits `watcher: FAILED - cycle ended without an actionable reason` and exits nonzero. -An attached arm follows verified identity-matched successors and reports the same typed failure if that chain ends without one. +A zero/empty child return rechecks the home lock and beacon, attaches to a verified healthy successor when one exists, or resolves the close against the watcher's bounded terminal-delivery ledger. +An attached arm follows verified identity-matched successors and resolves the same way when that chain ends without one, because it holds no handle on the watcher's stdout and cannot read the reason line itself. +Before releasing its singleton lock after printing an actionable reason, the watcher records that reason with its PID and process identity in `state/.watch-deliveries.log`. +A matching PID and identity lets an attached arm report the delivered reason and exit zero even after the durable wake queue was drained, while an unrelated queue producer or a recycled PID cannot satisfy the match. +Only a cycle with no matching delivery record emits `watcher: FAILED - cycle ended without an actionable reason` and exits nonzero. The arm layer appends one tab-separated record per observed cycle to `state/.watch-cycle-exits.log`. Each record includes arm and watcher PIDs, start and end timestamps, exit code and signal, classified reason, beacon age, lock identity before and after close, and successor disposition. diff --git a/tests/fm-session-lock-ancestry.test.sh b/tests/fm-session-lock-ancestry.test.sh new file mode 100755 index 00000000000..2f2e5094a47 --- /dev/null +++ b/tests/fm-session-lock-ancestry.test.sh @@ -0,0 +1,363 @@ +#!/usr/bin/env bash +# tests/fm-session-lock-ancestry.test.sh - session-lock harness identity +# (bin/fm-session-lock-lib.sh). +# +# Two layers. The unit cases drive the library's own functions behind a +# deterministic fake ps, so both platforms' reporting semantics are covered from +# either host: macOS reports argv[0] in `ps -o comm=`, while procps on Linux +# reports the kernel exec name and ignores argv[0] entirely. The end-to-end cases +# run the REAL Stop auto-arm inside real process trees whose shapes differ only +# in how the per-session process is named and what its parent is. Those trees are +# orphaned before the hook fires, so the ancestry walk terminates inside the +# fixture and can never escape into the session running this suite. +# shellcheck disable=SC2016 # single quotes are deliberate: $FM_HOME and $$ expand inside the fixture child +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +TMP_ROOT=$(fm_test_tmproot fm-session-lock-ancestry) +fm_git_identity fmtest fmtest@example.invalid + +LIB="$ROOT/bin/fm-session-lock-lib.sh" + +# Claude Code's native installer names the per-session executable by its version, +# so the harness identity has to survive a basename that says nothing. +CLAUDE_VERSION_DIR="$TMP_ROOT/claude-install/share/claude/versions" +mkdir -p "$CLAUDE_VERSION_DIR" +ln -s /bin/bash "$CLAUDE_VERSION_DIR/2.1.220" +VERSIONED_CLAUDE="$CLAUDE_VERSION_DIR/2.1.220" + +FAKEBIN=$(fm_fakebin "$TMP_ROOT/harness-bin") +ln -s /bin/bash "$FAKEBIN/claude" +NAMED_CLAUDE="$FAKEBIN/claude" + +# --- unit layer: identity behind a deterministic process table --------------- + +# Run one library expression with shadowing ps. kill is stubbed so +# liveness questions are decided by the process table alone. +lib_eval() { # + local fakebin=$1 expr=$2 + PATH="$fakebin:$PATH" bash -c " + . \"\$0\" + kill() { return 0; } + $expr + " "$LIB" +} + +test_version_named_session_is_identified_on_both_platforms() { + local dir fakebin shape got + dir="$TMP_ROOT/version-named" + fakebin=$(fm_fakebin "$dir") + mkdir -p "$dir/state" + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +set -u +field= pid= +while [ "$#" -gt 0 ]; do + case "$1" in + -o) field=$2; shift 2 ;; + -p) pid=$2; shift 2 ;; + *) shift ;; + esac +done +case "$pid:$field:${FM_TEST_CLAUDE_SHAPE:-linux}" in + 700:comm=:linux) printf '%s\n' '2.1.220' ;; + 700:args=:linux) printf '%s\n' '/opt/claude/versions/2.1.220 --resume' ;; + 700:comm=:macos) printf '%s\n' '/Users/u/.local/share/claude/versions/2.1.220' ;; + 700:args=:macos) printf '%s\n' '/Users/u/.local/share/claude/versions/2.1.220 --resume' ;; + 700:ppid=:*) printf '%s\n' 1 ;; + *:comm=:*) printf '%s\n' bash ;; + *:args=:*) printf '%s\n' 'bash /repo/bin/fm-claude-stop-autoarm.sh' ;; + *:ppid=:*) printf '%s\n' 700 ;; +esac +SH + chmod +x "$fakebin/ps" + printf '700\n' > "$dir/state/.lock" + + for shape in linux macos; do + got=$(FM_TEST_CLAUDE_SHAPE="$shape" lib_eval "$fakebin" 'fm_harness_ancestry_pid') \ + || fail "$shape: the version-named session was not found in the ancestry at all" + [ "$got" = 700 ] || fail "$shape: ancestry resolved '$got', expected the version-named session pid 700" + FM_TEST_CLAUDE_SHAPE="$shape" lib_eval "$fakebin" 'fm_harness_pid_alive 700' \ + || fail "$shape: a live version-named session was not recognized as a harness" + FM_TEST_CLAUDE_SHAPE="$shape" lib_eval "$fakebin" "fm_session_lock_owned_by_self '$dir/state'" \ + || fail "$shape: the session holding the lock did not recognize itself as the owner" + done + pass "session-lock: a version-named Claude Code session is identified from its install path and argv[0]" +} + +test_ordinary_paths_are_never_harness_processes() { + local dir fakebin shape + dir="$TMP_ROOT/ordinary-paths" + fakebin=$(fm_fakebin "$dir") + mkdir -p "$dir/state" + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +set -u +field= pid= +while [ "$#" -gt 0 ]; do + case "$1" in + -o) field=$2; shift 2 ;; + -p) pid=$2; shift 2 ;; + *) shift ;; + esac +done +case "$pid:$field:${FM_TEST_PATH_SHAPE:-hookdir}" in + 810:comm=:hookdir) printf '%s\n' '/home/u/.claude/hooks/notify.sh' ;; + 810:args=:hookdir) printf '%s\n' '/home/u/.claude/hooks/notify.sh --quiet' ;; + 810:comm=:piprefix) printf '%s\n' '/opt/pipeline/bin/runner' ;; + 810:args=:piprefix) printf '%s\n' '/opt/pipeline/bin/runner --once' ;; + 810:ppid=:*) printf '%s\n' 1 ;; + *:comm=:*) printf '%s\n' bash ;; + *:args=:*) printf '%s\n' 'bash /repo/bin/fm-watch-arm.sh' ;; + *:ppid=:*) printf '%s\n' 810 ;; +esac +SH + chmod +x "$fakebin/ps" + printf '810\n' > "$dir/state/.lock" + + # Identity may be read from an executable path, but only from whole path + # components: anything merely living under ~/.claude, and any component that + # merely starts with a harness name, must stay outside the harness identity. + for shape in hookdir piprefix; do + if FM_TEST_PATH_SHAPE="$shape" lib_eval "$fakebin" 'fm_harness_ancestry_pid'; then + fail "$shape: an ordinary script path was treated as a harness process" + fi + if FM_TEST_PATH_SHAPE="$shape" lib_eval "$fakebin" 'fm_harness_pid_alive 810'; then + fail "$shape: an ordinary script path passed the harness-liveness predicate" + fi + if FM_TEST_PATH_SHAPE="$shape" lib_eval "$fakebin" "fm_session_lock_owned_by_self '$dir/state'"; then + fail "$shape: an ordinary script path claimed the home's session lock" + fi + done + pass "session-lock: ordinary script paths under a harness directory are not harness processes" +} + +test_harness_beyond_a_gap_never_owns_the_lock() { + local dir fakebin got + dir="$TMP_ROOT/gap" + fakebin=$(fm_fakebin "$dir") + mkdir -p "$dir/state" + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +set -u +field= pid= +while [ "$#" -gt 0 ]; do + case "$1" in + -o) field=$2; shift 2 ;; + -p) pid=$2; shift 2 ;; + *) shift ;; + esac +done +case "$pid:$field" in + 900:comm=) printf '%s\n' claude ;; + 900:args=) printf '%s\n' 'claude' ;; + 900:ppid=) printf '%s\n' 910 ;; + 910:comm=) printf '%s\n' bash ;; + 910:args=) printf '%s\n' 'bash tests/run.sh' ;; + 910:ppid=) printf '%s\n' 920 ;; + 920:comm=) printf '%s\n' claude ;; + 920:args=) printf '%s\n' 'claude' ;; + 920:ppid=) printf '%s\n' 1 ;; + *:comm=) printf '%s\n' bash ;; + *:args=) printf '%s\n' bash ;; + *:ppid=) printf '%s\n' 900 ;; +esac +SH + chmod +x "$fakebin/ps" + + got=$(lib_eval "$fakebin" 'fm_harness_ancestry_pid') || fail "the contiguous harness run was not resolved" + [ "$got" = 900 ] || fail "ancestry crossed a non-harness gap, resolved '$got' instead of 900" + printf '920\n' > "$dir/state/.lock" + if lib_eval "$fakebin" "fm_session_lock_owned_by_self '$dir/state'"; then + fail "an unrelated harness beyond a non-harness gap was accepted as this session's lock owner" + fi + printf '900\n' > "$dir/state/.lock" + lib_eval "$fakebin" "fm_session_lock_owned_by_self '$dir/state'" \ + || fail "the contiguous harness run did not recognize its own lock" + pass "session-lock: ownership stops at the first non-harness gap above the contiguous run" +} + +test_competing_version_named_session_is_seen_as_live() { + local dir fakebin + dir="$TMP_ROOT/competing" + fakebin=$(fm_fakebin "$dir") + mkdir -p "$dir/state" + cat > "$fakebin/ps" <<'SH' +#!/usr/bin/env bash +set -u +field= pid= +while [ "$#" -gt 0 ]; do + case "$1" in + -o) field=$2; shift 2 ;; + -p) pid=$2; shift 2 ;; + *) shift ;; + esac +done +case "$pid:$field" in + 600:comm=) printf '%s\n' '2.1.220' ;; + 600:args=) printf '%s\n' '/opt/claude/versions/2.1.220' ;; + 600:ppid=) printf '%s\n' 1 ;; + 650:comm=) printf '%s\n' claude ;; + 650:args=) printf '%s\n' claude ;; + 650:ppid=) printf '%s\n' 1 ;; + *:comm=) printf '%s\n' bash ;; + *:args=) printf '%s\n' bash ;; + *:ppid=) printf '%s\n' 650 ;; +esac +SH + chmod +x "$fakebin/ps" + # pid 600 is a different live session that holds the lock; this process + # descends from 650 instead. Treating 600 as dead would let this session + # reclaim a live competitor's home. + printf '600\n' > "$dir/state/.lock" + if lib_eval "$fakebin" "fm_session_lock_owned_by_self '$dir/state'"; then + fail "a lock held outside this ancestry was claimed as this session's own" + fi + lib_eval "$fakebin" 'fm_harness_pid_alive 600' \ + || fail "a live competing version-named session was classified as a dead lock owner" + pass "session-lock: a live version-named session holding the lock is not mistaken for a stale owner" +} + +# --- end-to-end layer: the real Stop auto-arm in real process trees ---------- + +install_autoarm_scripts() { + local dir=$1 + mkdir -p "$dir/bin" + cp "$ROOT/bin/fm-claude-stop-autoarm.sh" "$dir/bin/fm-claude-stop-autoarm.sh" + cp "$ROOT/bin/fm-primary-scope-lib.sh" "$dir/bin/fm-primary-scope-lib.sh" + cp "$ROOT/bin/fm-supervision-lib.sh" "$dir/bin/fm-supervision-lib.sh" + cp "$ROOT/bin/fm-wake-lib.sh" "$dir/bin/fm-wake-lib.sh" + cp "$ROOT/bin/fm-session-lock-lib.sh" "$dir/bin/fm-session-lock-lib.sh" + cp "$ROOT/bin/fm-lock.sh" "$dir/bin/fm-lock.sh" + chmod +x "$dir/bin/fm-claude-stop-autoarm.sh" "$dir/bin/fm-lock.sh" + cat > "$dir/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +echo "$$" >> "$FM_HOME/state/arm-ran" +printf 'watcher: started pid=%s (beacon fresh)\n' "$$" +printf 'stale: fixture-win actionable\n' +exit 0 +SH + chmod +x "$dir/bin/fm-watch-arm.sh" +} + +# A primary home with one task in flight, so the hook's scope and supervision-need +# gates both pass and only identity decides the outcome. +make_primary_home() { # + local dir=$1 + mkdir -p "$dir/state" + git init -q "$dir" + git -C "$dir" commit -q --allow-empty -m init + : > "$dir/AGENTS.md" + : > "$dir/state/task.meta" + install_autoarm_scripts "$dir" + # The process that fires the hook records its own pid as the session lock + # owner, exactly as a real session does at session start. + cat > "$dir/session.sh" <<'SH' +#!/usr/bin/env bash +if [ "${FM_FIXTURE_ORPHAN_HERE:-0}" = 1 ]; then + i=0 + while [ "$i" -lt 200 ] && [ "$(ps -o ppid= -p $$ 2>/dev/null | tr -d ' ')" != 1 ]; do + sleep 0.05 + i=$((i + 1)) + done +fi +printf '%s\n' "$$" > "$FM_HOME/state/session-pid" +printf '%s\n' "$$" > "$FM_HOME/state/.lock" +"$FM_HOME/bin/fm-claude-stop-autoarm.sh" "$FM_HOME/state/hook.out" 2>&1 +printf '%s\n' "$?" > "$FM_HOME/state/hook.rc" +SH + cat > "$dir/daemon.sh" <<'SH' +#!/usr/bin/env bash +i=0 +while [ "$i" -lt 200 ] && [ "$(ps -o ppid= -p $$ 2>/dev/null | tr -d ' ')" != 1 ]; do + sleep 0.05 + i=$((i + 1)) +done +printf '%s\n' "$$" > "$FM_HOME/state/daemon-pid" +"$FM_SESSION_BIN" "$FM_HOME/session.sh" +exit 0 +SH + chmod +x "$dir/session.sh" "$dir/daemon.sh" +} + +# Start the fixture tree detached from this suite's own process tree: the +# launcher exits immediately, so the tree is reparented to init and the ancestry +# walk terminates inside the fixture. Returns once the hook has recorded its exit +# code. +run_fixture_tree() { # [] + local dir=$1 session_bin=$2 daemon_bin=${3:-} i + if [ -n "$daemon_bin" ]; then + FM_HOME="$dir" FM_SESSION_BIN="$session_bin" FM_FIXTURE_ORPHAN_HERE=0 \ + bash -c '"$0" "$1" &' "$daemon_bin" "$dir/daemon.sh" + else + FM_HOME="$dir" FM_FIXTURE_ORPHAN_HERE=1 \ + bash -c '"$0" "$1" &' "$session_bin" "$dir/session.sh" + fi + i=0 + while [ "$i" -lt 400 ] && [ ! -s "$dir/state/hook.rc" ]; do + sleep 0.05 + i=$((i + 1)) + done + [ -s "$dir/state/hook.rc" ] || fail "the fixture hook never finished" +} + +hook_rc() { + tr -d '[:space:]' < "$1/state/hook.rc" +} + +epoch_outcome() { + sed -n 's/^.*outcome=\([a-z][a-z]*\) .*$/\1/p' "$1/state/.claude-autoarm-epoch" 2>/dev/null || true +} + +test_e2e_version_named_session_claims_the_home() { + local dir + dir="$TMP_ROOT/e2e-version-named" + make_primary_home "$dir" + run_fixture_tree "$dir" "$VERSIONED_CLAUDE" + expect_code 2 "$(hook_rc "$dir")" "a version-named session must claim its home and rewake" + [ -e "$dir/state/arm-ran" ] || fail "supervision never armed for a version-named session" + [ "$(epoch_outcome "$dir")" = rewake ] || fail "no claim was recorded, got: $(epoch_outcome "$dir")" + pass "session-lock e2e: a version-named session claims the home and arms supervision" +} + +test_e2e_daemon_parented_session_claims_the_home() { + local dir session_pid daemon_pid lock_after + dir="$TMP_ROOT/e2e-daemon-parented" + make_primary_home "$dir" + run_fixture_tree "$dir" "$NAMED_CLAUDE" "$NAMED_CLAUDE" + session_pid=$(tr -d '[:space:]' < "$dir/state/session-pid") + daemon_pid=$(tr -d '[:space:]' < "$dir/state/daemon-pid") + [ -n "$session_pid" ] && [ "$session_pid" != "$daemon_pid" ] \ + || fail "fixture did not produce a distinct daemon and session: session=$session_pid daemon=$daemon_pid" + lock_after=$(tr -d '[:space:]' < "$dir/state/.lock") + expect_code 2 "$(hook_rc "$dir")" "a session parented by a harness-named daemon must claim its home and rewake" + [ -e "$dir/state/arm-ran" ] || fail "supervision never armed for a daemon-parented session" + [ "$lock_after" = "$session_pid" ] || fail "the session lock moved off the session: expected $session_pid, got $lock_after" + pass "session-lock e2e: a session parented by a harness-named daemon claims the home and arms supervision" +} + +test_e2e_daemon_parented_version_named_session_keeps_its_lock() { + local dir session_pid daemon_pid lock_after + dir="$TMP_ROOT/e2e-daemon-version-named" + make_primary_home "$dir" + run_fixture_tree "$dir" "$VERSIONED_CLAUDE" "$NAMED_CLAUDE" + session_pid=$(tr -d '[:space:]' < "$dir/state/session-pid") + daemon_pid=$(tr -d '[:space:]' < "$dir/state/daemon-pid") + lock_after=$(tr -d '[:space:]' < "$dir/state/.lock") + [ "$lock_after" != "$daemon_pid" ] \ + || fail "the live session's lock was reclaimed as stale and rewritten to the shared daemon pid $daemon_pid" + [ "$lock_after" = "$session_pid" ] || fail "the session lock moved off the session: expected $session_pid, got $lock_after" + expect_code 2 "$(hook_rc "$dir")" "a version-named session under a daemon must claim its home and rewake" + [ -e "$dir/state/arm-ran" ] || fail "supervision never armed for a version-named daemon-parented session" + pass "session-lock e2e: a version-named session under a harness-named daemon keeps its own lock" +} + +test_version_named_session_is_identified_on_both_platforms +test_ordinary_paths_are_never_harness_processes +test_harness_beyond_a_gap_never_owns_the_lock +test_competing_version_named_session_is_seen_as_live +test_e2e_version_named_session_claims_the_home +test_e2e_daemon_parented_session_claims_the_home +test_e2e_daemon_parented_version_named_session_keeps_its_lock diff --git a/tests/fm-watch-arm.test.sh b/tests/fm-watch-arm.test.sh new file mode 100755 index 00000000000..9540e919c6f --- /dev/null +++ b/tests/fm-watch-arm.test.sh @@ -0,0 +1,151 @@ +#!/usr/bin/env bash +# tests/fm-watch-arm.test.sh - the arm layer's cycle-close contract when the arm +# did not own the cycle. +# +# The watcher prints its one reason line to its OWN stdout, so only the arm that +# forked it ever reads that line. An arm that ATTACHED to an existing cycle holds +# no handle on it and can observe only a released lock, which is why a completely +# successful cycle used to be reported as +# "watcher: FAILED - cycle ended without an actionable reason" on every harness +# whose protocol reads that line. These are real-process tests: a real +# bin/fm-watch.sh holds the singleton, a real bin/fm-watch-arm.sh attaches to it, +# and a real status change drives a real wake through the watcher-bound delivery +# record and durable queue. +set -u + +# shellcheck source=tests/wake-helpers.sh +. "$(dirname "${BASH_SOURCE[0]}")/wake-helpers.sh" + +WATCH="$ROOT/bin/fm-watch.sh" +WATCH_ARM="$ROOT/bin/fm-watch-arm.sh" +DRAIN="$ROOT/bin/fm-wake-drain.sh" + +TMP_ROOT=$(fm_test_tmproot fm-watch-arm-tests) + +# Both starters background a real process the test later waits on, so they set a +# global instead of echoing: a command substitution would make the pid a child of +# a subshell this shell can no longer wait for. +SEED_PID= +ARM_PID= + +# Start the real watcher as the singleton holder. +start_seed_watcher() { # + local state=$1 fakebin=$2 out=$3 i + PATH="$fakebin:$PATH" FM_STATE_OVERRIDE="$state" FM_POLL=5 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + SEED_PID=$! + i=0 + while [ "$i" -lt 60 ]; do + [ "$(cat "$state/.watch.lock/pid" 2>/dev/null || true)" = "$SEED_PID" ] \ + && [ -e "$state/.last-watcher-beat" ] && break + sleep 0.1 + i=$((i + 1)) + done + [ "$(cat "$state/.watch.lock/pid" 2>/dev/null || true)" = "$SEED_PID" ] \ + || fail "seed watcher did not take the lock" +} + +# Attach a real arm to the live cycle. +start_attached_arm() { # + local state=$1 fakebin=$2 armout=$3 confirm=$4 i + PATH="$fakebin:$PATH" FM_STATE_OVERRIDE="$state" FM_ARM_ATTACH_POLL=0.1 \ + FM_ARM_CONFIRM_TIMEOUT="$confirm" "$WATCH_ARM" > "$armout" & + ARM_PID=$! + i=0 + while [ "$i" -lt 80 ]; do + grep -qF "watcher: attached pid=$SEED_PID" "$armout" 2>/dev/null && break + sleep 0.1 + i=$((i + 1)) + done + grep -qF "watcher: attached pid=$SEED_PID" "$armout" \ + || fail "arm did not attach to the live watcher: $(cat "$armout")" +} + +test_attached_arm_reports_the_delivered_wake() { + local dir state fakebin out armout status + dir=$(make_case attached-delivered-wake) + state="$dir/state" + fakebin="$dir/fakebin" + out="$dir/watch.out" + armout="$dir/arm.out" + start_seed_watcher "$state" "$fakebin" "$out" + start_attached_arm "$state" "$fakebin" "$armout" 1 + + # A real captain-relevant status change: the watcher records it in the durable + # queue, prints its one reason line to its own stdout, and exits. + printf 'done: fixture finished\n' > "$state/demo.status" + wait_for_exit "$SEED_PID" 120 + grep -q '^signal:' "$out" || fail "seed watcher did not surface the signal wake: $(cat "$out")" + + wait_for_exit "$ARM_PID" 120 + status=$? + grep -q 'demo.status' "$state/.wake-queue" \ + || fail "the wake was not durably recorded, so this case proves nothing" + ! grep -qF 'watcher: FAILED' "$armout" \ + || fail "attached arm reported a delivered wake as a failed cycle: $(cat "$armout")" + grep -q '^signal:' "$armout" \ + || fail "attached arm did not report the durably recorded wake reason: $(cat "$armout")" + expect_code 0 "$status" "an attached arm whose cycle delivered a wake must close successfully" + grep -q 'reason=attached-delivered-wake' "$state/.watch-cycle-exits.log" \ + || fail "the delivered-wake close was not classified in the lifecycle ledger" + pass "watch-arm: an attached arm reports the wake its cycle delivered instead of a false failure" +} + +test_attached_arm_reports_the_delivered_wake_after_drain() { + local dir state fakebin out armout status + dir=$(make_case attached-drained-wake) + state="$dir/state" + fakebin="$dir/fakebin" + out="$dir/watch.out" + armout="$dir/arm.out" + start_seed_watcher "$state" "$fakebin" "$out" + # A wider confirmation budget keeps the arm in its successor wait while the + # handling turn drains, which is the ordering this case exists to cover. + start_attached_arm "$state" "$fakebin" "$armout" 5 + + printf 'done: fixture finished\n' > "$state/demo.status" + wait_for_exit "$SEED_PID" 120 + # The handling turn consumes the records before the attached arm closes: the + # queue is empty again, while the watcher's identity-bound terminal record + # still proves which cycle delivered the reason. + FM_STATE_OVERRIDE="$state" "$DRAIN" >/dev/null 2>&1 || fail "drain failed" + [ ! -s "$state/.wake-queue" ] || fail "drain left records behind" + + wait_for_exit "$ARM_PID" 200 + status=$? + ! grep -qF 'watcher: FAILED' "$armout" \ + || fail "attached arm reported an already-handled wake as a failed cycle: $(cat "$armout")" + grep -q '^signal:' "$armout" \ + || fail "attached arm did not report the delivered reason after the queue drain: $(cat "$armout")" + expect_code 0 "$status" "an attached arm whose wake was already drained must close successfully" + pass "watch-arm: a delivered wake consumed by the handling turn still closes the attached arm cleanly" +} + +test_attached_arm_still_fails_on_a_wake_it_did_not_deliver() { + local dir state fakebin out armout status + dir=$(make_case attached-no-delivery) + state="$dir/state" + fakebin="$dir/fakebin" + out="$dir/watch.out" + armout="$dir/arm.out" + start_seed_watcher "$state" "$fakebin" "$out" + start_attached_arm "$state" "$fakebin" "$armout" 1 + + # A process-event producer advances the same home-wide queue while the + # observed watcher remains uninvolved, so only watcher-bound evidence can + # distinguish this from a delivered watcher cycle. + append_wake "$state" check process-event "check: process-event result captured: fixture" + kill "$SEED_PID" 2>/dev/null || true + wait "$SEED_PID" 2>/dev/null || true + wait_for_exit "$ARM_PID" 120 + status=$? + grep -qF 'watcher: FAILED - cycle ended without an actionable reason' "$armout" \ + || fail "a cycle that delivered nothing must still fail loudly: $(cat "$armout")" + [ "$status" -ne 0 ] && [ "$status" -ne 124 ] \ + || fail "arm did not exit nonzero for a cycle that delivered nothing (status $status)" + pass "watch-arm: a cycle that delivered no wake of its own still fails loudly" +} + +test_attached_arm_reports_the_delivered_wake +test_attached_arm_reports_the_delivered_wake_after_drain +test_attached_arm_still_fails_on_a_wake_it_did_not_deliver From 33a428773059aa9bbe55865ce73ae51e4334bccc Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Sun, 2 Aug 2026 15:53:17 -0700 Subject: [PATCH 02/35] fix(bin): harden Claude supervision auto-arm recovery (#1495) * fix(supervision): harden Claude auto-arm failure handling * no-mistakes(review): Guarantee automatic retry after Claude auto-arm failures * no-mistakes(review): Gate attended fail-open on verified supervision failure * no-mistakes(document): Document Claude auto-arm retry and guard scope * no-mistakes: apply CI fixes * fix(supervision): make Claude fail-open progression monotonic * no-mistakes(review): Preserve auto-arm failure episodes until verified watcher recovery * no-mistakes(review): Linearize auto-arm failure progression across existing locks * no-mistakes(review): Linearize positive recovery across shared failure episode lock * no-mistakes(review): Scope Claude recovery contention to Claude guard mode * no-mistakes(document): Align supervision auto-arm documentation * no-mistakes(review): Preserve actionable wakes despite healthy successors * no-mistakes(document): Refresh supervision auto-arm documentation --- AGENTS.md | 2 +- bin/fm-claude-stop-autoarm.sh | 143 +++++-- bin/fm-guard.sh | 28 +- bin/fm-supervision-instructions.sh | 2 +- bin/fm-supervision-lib.sh | 8 +- bin/fm-turnend-guard.sh | 212 ++++++++-- bin/fm-wake-lib.sh | 56 +++ docs/architecture.md | 7 +- docs/configuration.md | 7 +- docs/supervision-protocols/claude.md | 13 +- docs/turnend-guard.md | 28 +- docs/verification/supervision.md | 40 ++ docs/watcher-continuity.md | 15 +- tests/fm-claude-stop-autoarm.test.sh | 186 ++++++++- tests/fm-guard-stale-banner.test.sh | 49 ++- tests/fm-secondmate-harness.test.sh | 14 + tests/fm-supervision-instructions.test.sh | 11 +- tests/fm-turnend-guard.test.sh | 461 +++++++++++++++++++++- tests/fm-wake-queue.test.sh | 18 +- tests/fm-watcher-lock.test.sh | 22 +- tests/fm-x-mode.test.sh | 3 +- 21 files changed, 1158 insertions(+), 167 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 975c5e793ce..6e8eea580ad 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -112,7 +112,7 @@ state/ volatile runtime signals; gitignored .wake-queue durable queued wakes: epochseqkindkeypayload .afk durable away-mode flag; present = sub-supervisor may inject escalations (set by /afk, cleared on user return) .watch.lock .wake-queue.lock watcher singleton and queue serialization locks - .claude-autoarm.lock .claude-autoarm-epoch .turnend-claude-blocks Claude Stop auto-arm single-flight, epoch, and guard-budget records; never touch + .claude-autoarm.lock .claude-autoarm-epoch .claude-autoarm-failure-notified .claude-autoarm-failure-alarmed .turnend-claude-blocks .turnend-claude-blocks.lock Claude Stop auto-arm single-flight, epoch, failure-episode, attended-alarm, guard-budget, and budget-lock records; never touch .hash-* .count-* .stale-* .stale-since-* .paused-* .wedge-escalations-* .seen-* .hb-surfaced-* .last-* .heartbeat-streak watcher internals; never touch .watch-triage.log watcher's absorbed-wake debug log (size-capped); never relied on, safe to delete .last-watcher-beat watcher liveness beacon, touched every poll (including while absorbing benign wakes); guard scripts read it diff --git a/bin/fm-claude-stop-autoarm.sh b/bin/fm-claude-stop-autoarm.sh index df9ee1128fc..c23098c4405 100755 --- a/bin/fm-claude-stop-autoarm.sh +++ b/bin/fm-claude-stop-autoarm.sh @@ -28,15 +28,24 @@ # this hook-owned process tree (never shell &); Claude owns the process # group, so its timeout/session teardown kills arm and watcher together. # - Translation: while supervision is still needed and AFK remains inactive, -# an actionable arm close (signal:/stale:/check:/heartbeat) or a typed -# watcher: FAILED prints one rewake banner to stderr and exits 2, which -# wakes Claude even while idle ("Stop hook feedback"). A clean close with -# no actionable reason and no remaining need exits 0 silently. +# an actionable arm close (signal:/stale:/check:/heartbeat) prints one +# rewake banner to stderr and exits 2, which wakes Claude even while idle +# ("Stop hook feedback"). A close that reports no actionable reason is +# benign when a live identity-matched watcher still has a fresh beacon. +# - Failure handling: a typed failure is rechecked against the same live, +# fresh watcher predicate and retried a bounded number of times in this +# hook. Only an exhausted failure with no verified watcher emits one +# last-resort notice per failure episode; later consecutive failures still +# exit 2 to guarantee the next Stop-owned retry without repeating notice, +# until the synchronous guard has consumed its attended fail-open. # # The epoch ledger state/.claude-autoarm-epoch records the latest claim and # outcome so the synchronous Stop guard (bin/fm-turnend-guard.sh --claude) can # allow a stop whose recovery this hook already owns, instead of forcing a -# duplicate continuation for the same event epoch. +# duplicate continuation for the same event epoch. The failure marker +# state/.claude-autoarm-failure-notified deduplicates the last-resort notice, +# and state/.claude-autoarm-failure-alarmed bounds the attended fail-open and +# suppresses any later automatic continuation in that unresolved episode. # # This hook never blocks the Stop decision itself and never prints to stdout: # exit 0 is always silent, and exit 2 carries the rewake banner on stderr. @@ -53,6 +62,13 @@ CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" GRACE=${FM_GUARD_GRACE:-300} OWNER_LOCK="$STATE/.claude-autoarm.lock" EPOCH="$STATE/.claude-autoarm-epoch" +FAILURE_NOTICE="$STATE/.claude-autoarm-failure-notified" +FAILURE_ALARM="$STATE/.claude-autoarm-failure-alarmed" +AUTOARM_ATTEMPTS=${FM_CLAUDE_AUTOARM_ATTEMPTS:-2} +case "$AUTOARM_ATTEMPTS" in + 1|2|3) : ;; + *) AUTOARM_ATTEMPTS=2 ;; +esac # shellcheck source=bin/fm-primary-scope-lib.sh . "$SCRIPT_DIR/fm-primary-scope-lib.sh" @@ -109,6 +125,10 @@ fi # owner foregrounds the arm and translates its close; every other firing exits # 0 so one watcher cycle maps to at most one exit-2 rewake. fm_lock_try_acquire "$OWNER_LOCK" || exit 0 +if ! fm_lock_set_role "$OWNER_LOCK" autoarm; then + fm_lock_release "$OWNER_LOCK" + exit 0 +fi trap 'fm_lock_release "$OWNER_LOCK"' EXIT write_epoch() { # @@ -136,59 +156,100 @@ write_epoch arming # NO shell &: this hook process tree is the harness-owned lifecycle. The arm # forks the watcher as its own tracked child exactly as it does for the # model-driven background-task path, and propagates the wake reason on close. -OUT=$(mktemp "$STATE/.claude-autoarm-output.XXXXXX") || OUT= -if [ -n "$OUT" ]; then - "$SCRIPT_DIR/fm-watch-arm.sh" >"$OUT" 2>&1 - RC=$? -else - "$SCRIPT_DIR/fm-watch-arm.sh" >/dev/null 2>&1 - RC=$? -fi +# Every non-actionable close is checked against the same identity-matched live +# watcher and fresh-beacon predicate used by the turn-end guard before it is +# retried or translated into an operator-visible failure. +OUT= +ACTIONABLE=0 +HEALTHY=0 +attempt=0 +while [ "$attempt" -lt "$AUTOARM_ATTEMPTS" ]; do + attempt=$((attempt + 1)) + OUT=$(mktemp "$STATE/.claude-autoarm-output.XXXXXX") || OUT= + if [ -n "$OUT" ]; then + "$SCRIPT_DIR/fm-watch-arm.sh" >"$OUT" 2>&1 || true + else + "$SCRIPT_DIR/fm-watch-arm.sh" >/dev/null 2>&1 || true + fi + + # AFK may have appeared mid-cycle: the daemon owns triage now, so suppress + # every subsequent classification and handoff. + if [ -e "$STATE/.afk" ]; then + write_epoch afk + [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true + exit 0 + fi + + ACTIONABLE=0 + if [ -n "$OUT" ]; then + grep -Eq '^(signal:|stale:|check:|heartbeat($|:))' "$OUT" 2>/dev/null && ACTIONABLE=1 + fi + [ "$ACTIONABLE" -eq 1 ] && break + + # A non-actionable close is benign when another verified watcher already owns + # this home and is still beating within the shared grace window. + if fm_watcher_healthy "$STATE" "$SCRIPT_DIR/fm-watch.sh" "$GRACE" "$FM_HOME"; then + HEALTHY=1 + break + fi + [ "$attempt" -lt "$AUTOARM_ATTEMPTS" ] || break + [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true + OUT= +done -# --- classify and translate --------------------------------------------------- -# AFK may have appeared mid-cycle: the daemon owns triage now, so suppress the -# rewake even for an actionable close. -if [ -e "$STATE/.afk" ]; then - write_epoch afk +# The need may have vanished mid-cycle (fleet torn down, X opted out): nothing +# left to supervise, so close quietly instead of waking the model. +if ! need_supervision; then + write_epoch clean [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true exit 0 fi -ACTIONABLE=0 -FAILED=0 -if [ -n "$OUT" ]; then - grep -Eq '^(signal:|stale:|check:|heartbeat($|:))' "$OUT" 2>/dev/null && ACTIONABLE=1 - grep -q '^watcher: FAILED' "$OUT" 2>/dev/null && FAILED=1 -fi -[ "$RC" -ne 0 ] && FAILED=1 - -if [ "$ACTIONABLE" -eq 0 ] && [ "$FAILED" -eq 0 ]; then - write_epoch clean +if [ "$HEALTHY" -eq 1 ]; then + if fm_failure_episode_reset "$STATE"; then + write_epoch clean + [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true + exit 0 + fi + write_epoch failed-suppressed [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true - exit 0 + [ -e "$FAILURE_ALARM" ] && exit 0 + exit 2 fi -# The need may have vanished mid-cycle (fleet torn down, X opted out): nothing -# left to supervise, so close quietly instead of waking the model. -if ! need_supervision; then - write_epoch clean +# After the synchronous guard has consumed the episode's attended fail-open, +# do not create another exit-2 continuation that could defeat it. +if [ -e "$FAILURE_ALARM" ]; then + write_epoch failed-suppressed [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true exit 0 fi -write_epoch rewake -if [ "$FAILED" -eq 1 ]; then - { - printf 'firstmate watcher cycle FAILED - supervision is down while this home still needs it.\n' - [ -n "$OUT" ] && grep -E '^(watcher:|signal:|stale:|check:|heartbeat)' "$OUT" 2>/dev/null | head -8 - printf 'Run bin/fm-wake-drain.sh first. Then repair supervision with bin/fm-watch-arm.sh as its own Claude Code background task (never shell &). If the failure repeats, treat it as a blocker and report it instead of ending blind.\n' - } >&2 -else +if [ "$ACTIONABLE" -eq 1 ]; then + write_epoch rewake { printf 'firstmate watcher wake - one supervision event needs a handling turn now.\n' [ -n "$OUT" ] && grep -E '^(signal:|stale:|check:|heartbeat)' "$OUT" 2>/dev/null | head -8 printf 'Run bin/fm-wake-drain.sh first and handle the wake. This Stop hook owns watcher continuity: when the handling turn ends, the next needed cycle arms automatically - do NOT run bin/fm-watch-arm.sh after an ordinary wake.\n' } >&2 + [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true + exit 2 +fi + +# Notify only once for this continuous failure episode; every later invocation +# still exits 2 so Claude must continue into another Stop-owned retry without +# creating a repeated operator notice or manual-arm loop. +if [ ! -e "$FAILURE_NOTICE" ]; then + write_epoch failed + { + printf 'firstmate watcher auto-arm FAILED - the Stop-owned automatic supervision mechanism is broken after %s bounded attempts, and no live watcher with a fresh beacon was verified.\n' "$attempt" + [ -n "$OUT" ] && grep -E '^(watcher:|signal:|stale:|check:|heartbeat)' "$OUT" 2>/dev/null | head -8 + printf 'Do not launch a manual background arm from this notice; investigate the automatic Stop hook and watcher startup before ending blind.\n' + } >&2 + : > "$FAILURE_NOTICE" 2>/dev/null || true + [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true + exit 2 fi +write_epoch failed-suppressed [ -z "$OUT" ] || rm -f "$OUT" 2>/dev/null || true exit 2 diff --git a/bin/fm-guard.sh b/bin/fm-guard.sh index 02df76e9543..5698d376ab0 100755 --- a/bin/fm-guard.sh +++ b/bin/fm-guard.sh @@ -5,9 +5,10 @@ # First, always warn if the firstmate primary checkout (FM_ROOT) is on a named # non-default branch, because that means firstmate-on-itself work landed in the # primary instead of an isolated worktree. -# Then, if any task is in flight (a state/.meta exists) and the watcher's -# liveness beacon (state/.last-watcher-beat, touched every poll cycle) is -# missing or older than FM_GUARD_GRACE seconds, prints a loud, clearly delimited +# Then, if a task is in flight (a state/.meta exists) or X-mode relay +# polling is active (state/x-watch.check.sh exists) and no identity-matched +# watcher has a liveness beacon (state/.last-watcher-beat, touched every poll +# cycle) fresh within FM_GUARD_GRACE seconds, prints a loud, clearly delimited # banner so the agent cannot skim past it in the tool output of whatever it was # doing - the one channel every harness has. The full banner is emitted once per # distinct staleness episode in this FM_HOME (keyed to beacon mtime or absence); @@ -24,6 +25,7 @@ FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" +WATCH="$SCRIPT_DIR/fm-watch.sh" GRACE=${FM_GUARD_GRACE:-300} queue_pending=false READ_ONLY=${FM_GUARD_READ_ONLY:-0} @@ -140,20 +142,22 @@ if [ -n "$tangle_branch" ]; then } >&2 fi -# Compute in-flight count and watcher-beacon freshness via the shared -# grace-based predicate (bin/fm-supervision-lib.sh). Only act with tasks in -# flight; count them so the banner can say how much is riding on an absent -# watcher. +# Compute supervision need and watcher-beacon freshness via the shared +# grace-based predicate (bin/fm-supervision-lib.sh). Act when work, an event +# source, or an X-mode relay poll needs supervision. fm_supervision_status "$STATE" "$GRACE" in_flight=$FM_SUP_IN_FLIGHT sources=$FM_SUP_SOURCES needed=$FM_SUP_NEEDED -watcher_fresh=$FM_SUP_WATCHER_FRESH beacon_desc=$FM_SUP_BEACON_DESC +watcher_healthy=false +if fm_watcher_healthy "$STATE" "$WATCH" "$GRACE" "$FM_HOME"; then + watcher_healthy=true +fi if [ "$needed" = false ]; then - # Leave the unhealthy state (no work riding on the watcher): clear so a later - # in-flight + stale combination is a fresh episode even if the beacon is still - # absent with the same key string. + # Leave the unhealthy state (nothing riding on the watcher): clear so a later + # work or X-mode need + stale combination is a fresh episode even if the + # beacon is still absent with the same key string. [ "$READ_ONLY" -eq 1 ] || fm_guard_clear_stale_banner exit 0 fi @@ -163,7 +167,7 @@ fi # No fresh watcher with tasks in flight is the dangerous state: emit a prominent, # bordered banner FIRST so it reads as an alarm, not a buried stderr line. Later # calls in the same episode get a one-line reminder only. -if [ "$watcher_fresh" = false ]; then +if [ "$watcher_healthy" = false ]; then episode_key=$(fm_guard_stale_episode_key "$STATE") episode_key=${episode_key%$'\n'} print_full_banner=0 diff --git a/bin/fm-supervision-instructions.sh b/bin/fm-supervision-instructions.sh index 6cd87699b0b..5906649a555 100755 --- a/bin/fm-supervision-instructions.sh +++ b/bin/fm-supervision-instructions.sh @@ -135,7 +135,7 @@ repair_line() { case "$HARNESS" in claude) - printf '%s%s\n' "$prefix" 'repair missing watcher supervision with bin/fm-watch-arm.sh as its own Claude Code background task, never shell &.' + printf '%s%s\n' "$prefix" 'watcher supervision needs Stop-owned automatic recovery; inspect the hook registration and startup status before ending the turn.' ;; codex) printf '%s%s%s%s\n' "$prefix" 'repair missing watcher supervision with a foreground checkpoint: bin/fm-watch-checkpoint.sh --seconds ' "$checkpoint_seconds" '.' diff --git a/bin/fm-supervision-lib.sh b/bin/fm-supervision-lib.sh index 623b9b6d861..dc223d344ce 100644 --- a/bin/fm-supervision-lib.sh +++ b/bin/fm-supervision-lib.sh @@ -6,10 +6,10 @@ # work (a state/.meta exists) or an X-mode relay poll # (state/x-watch.check.sh), and whether its watcher has a fresh liveness beacon # (state/.last-watcher-beat, touched every poll cycle, within the grace window). -# bin/fm-guard.sh keeps its task-specific grace-based warning predicate; -# bin/fm-turnend-guard.sh uses the status fields here for its banner but performs -# its end-of-turn block decision with the live watcher lock check in -# bin/fm-wake-lib.sh. +# bin/fm-guard.sh and bin/fm-turnend-guard.sh use fm_watcher_healthy from +# bin/fm-wake-lib.sh for their warning and block decisions, so a fresh leftover +# beacon never counts as a live watcher. The status fields here retain the +# beacon-age details used in their messages. # Portable mtime; Linux stat lacks -f, macOS stat lacks -c. fm_sup_stat_mtime() { diff --git a/bin/fm-turnend-guard.sh b/bin/fm-turnend-guard.sh index a9ea3e5b0e5..dcd7a8ff9bc 100755 --- a/bin/fm-turnend-guard.sh +++ b/bin/fm-turnend-guard.sh @@ -48,15 +48,16 @@ # 1. a live identity-matched watcher with a fresh beacon allows immediately; # 2. otherwise wait briefly (FM_CLAUDE_AUTOARM_SYNC_WAIT_MS, default 800ms) # for the auto-arm to claim this home (state/.claude-autoarm.lock owner -# alive) or to record a fresh rewake outcome (state/.claude-autoarm-epoch) -# for this event epoch - either proof allows without consuming a -# continuation, so one event epoch yields exactly one recovery turn; +# alive) or to record a fresh actionable exit-2 outcome +# (state/.claude-autoarm-epoch) for this event epoch - either proof allows +# without consuming a continuation, so one event epoch yields exactly one recovery turn; +# the first fresh exhausted-failure epoch preserves the bounded progression, +# while later fresh failed epochs consume it instead of resetting it; # 3. only when neither materializes is the auto-arm genuinely absent: re-block # with the repair banner, bounded to FM_CLAUDE_TURNEND_BLOCK_BUDGET # (default 3) consecutive blocks per session - safely below Claude Code's -# hard 8-consecutive-block override - then allow degraded with a visible -# systemMessage so the session can always end. -# Any allow resets the consecutive-block budget. +# hard 8-consecutive-block override - then allow one loud attended +# fail-open only for an already verified failure episode. set -u SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -128,19 +129,27 @@ fm_primary_scope_matches "$FM_ROOT" "$STATE" || exit 0 . "$SCRIPT_DIR/fm-wake-lib.sh" BUDGET_FILE="$STATE/.turnend-claude-blocks" +BUDGET_LOCK="$STATE/.turnend-claude-blocks.lock" +OWNER_LOCK="$STATE/.claude-autoarm.lock" +FAILURE_NOTICE="$STATE/.claude-autoarm-failure-notified" +FAILURE_ALARM="$STATE/.claude-autoarm-failure-alarmed" +SESSION_ID=$(printf '%s' "$PAYLOAD" | jq -r '.session_id // "unknown"' 2>/dev/null || printf 'unknown') budget_reset() { [ "$CLAUDE_MODE" -eq 1 ] || return 0 + fm_lock_try_acquire "$BUDGET_LOCK" || return 0 rm -f "$BUDGET_FILE" 2>/dev/null || true + fm_lock_release "$BUDGET_LOCK" } fm_supervision_status "$STATE" "$GRACE" if [ "$FM_SUP_NEEDED" = false ]; then - budget_reset + [ -e "$FAILURE_NOTICE" ] || budget_reset exit 0 fi if fm_watcher_healthy "$STATE" "$WATCH" "$GRACE" "$FM_HOME"; then - budget_reset - exit 0 + [ "$CLAUDE_MODE" -eq 1 ] || exit 0 + fm_failure_episode_reset "$STATE" && exit 0 + exit 2 fi block_stop() { @@ -179,49 +188,180 @@ fi # The Stop-owned auto-arm fires on the same Stop event. Give it a brief bounded # window to prove it owns recovery for this event epoch before consuming one of # Claude's bounded continuations. +budget_account_current_epoch() { + local current_epoch outcome old_session old_count old_epoch tmp initialized + fm_lock_try_acquire "$BUDGET_LOCK" || return 1 + current_epoch=$(sed -n 's/^epoch=\([0-9][0-9]*\) .*/\1/p' "$STATE/.claude-autoarm-epoch" 2>/dev/null || true) + outcome=$(sed -n 's/^.*outcome=\([a-z][a-z-]*\) .*$/\1/p' "$STATE/.claude-autoarm-epoch" 2>/dev/null || true) + initialized=0 + COUNT=0 + if [ -f "$BUDGET_FILE" ]; then + old_session=$(sed -n '1s/^session=//p' "$BUDGET_FILE" 2>/dev/null || true) + old_count=$(sed -n '2s/^count=//p' "$BUDGET_FILE" 2>/dev/null || true) + old_epoch=$(sed -n '3s/^epoch=//p' "$BUDGET_FILE" 2>/dev/null || true) + case "$old_count" in + ''|*[!0-9]*) old_count=0 ;; + esac + if [ "$old_session" = "$SESSION_ID" ]; then + COUNT=$old_count + if [ -n "$current_epoch" ] && [ "$old_epoch" = "$current_epoch" ]; then + : + else + COUNT=$((COUNT + 1)) + fi + fi + fi + if [ ! -f "$BUDGET_FILE" ] || [ "${old_session:-}" != "$SESSION_ID" ]; then + case "$outcome" in + failed|failed-suppressed) + if [ -e "$FAILURE_NOTICE" ]; then + initialized=1 + COUNT=0 + else + COUNT=1 + fi + ;; + *) COUNT=1 ;; + esac + fi + tmp="$BUDGET_FILE.tmp.$$" + if ! printf 'session=%s\ncount=%s\nepoch=%s\n' "$SESSION_ID" "$COUNT" "$current_epoch" > "$tmp" 2>/dev/null \ + || ! mv -f "$tmp" "$BUDGET_FILE" 2>/dev/null; then + rm -f "$tmp" 2>/dev/null || true + fm_lock_release "$BUDGET_LOCK" + return 1 + fi + rm -f "$tmp" 2>/dev/null || true + BUDGET_INITIALIZED_FAILURE=$initialized + fm_lock_release "$BUDGET_LOCK" + return 0 +} + autoarm_owns_recovery() { - local pid outcome age + local pid role outcome age fm_watcher_healthy "$STATE" "$WATCH" "$GRACE" "$FM_HOME" && return 0 - pid=$(cat "$STATE/.claude-autoarm.lock/pid" 2>/dev/null || true) - fm_pid_alive "$pid" && return 0 - outcome=$(sed -n 's/^.*outcome=\([a-z][a-z]*\) .*$/\1/p' "$STATE/.claude-autoarm-epoch" 2>/dev/null || true) - if [ "$outcome" = rewake ]; then - age=$(fm_path_age "$STATE/.claude-autoarm-epoch") - [ "$age" -lt "$EPOCH_FRESH" ] && return 0 + pid=$(cat "$OWNER_LOCK/pid" 2>/dev/null || true) + role=$(fm_lock_role "$OWNER_LOCK" 2>/dev/null || true) + if fm_pid_alive "$pid" && [ "$role" = autoarm ]; then + [ ! -e "$FAILURE_NOTICE" ] || budget_account_current_epoch || true + return 0 fi + outcome=$(sed -n 's/^.*outcome=\([a-z][a-z-]*\) .*$/\1/p' "$STATE/.claude-autoarm-epoch" 2>/dev/null || true) + case "$outcome" in + rewake) + age=$(fm_path_age "$STATE/.claude-autoarm-epoch") + if [ "$age" -lt "$EPOCH_FRESH" ]; then + [ ! -e "$FAILURE_NOTICE" ] || budget_account_current_epoch || true + return 0 + fi + ;; + failed) + age=$(fm_path_age "$STATE/.claude-autoarm-epoch") + if [ "$age" -lt "$EPOCH_FRESH" ] && [ -e "$FAILURE_NOTICE" ] \ + && budget_account_current_epoch; then + [ "$BUDGET_INITIALIZED_FAILURE" -eq 1 ] && return 0 + fi + ;; + failed-suppressed) + age=$(fm_path_age "$STATE/.claude-autoarm-epoch") + if [ "$age" -lt "$EPOCH_FRESH" ] && [ -e "$FAILURE_NOTICE" ] \ + && budget_account_current_epoch; then + : + fi + ;; + esac return 1 } +terminal_fail_open() { + local pid role old_session old_count + [ "$COUNT" -gt "$BLOCK_BUDGET" ] || return 1 + failure_episode_verified || return 1 + [ ! -e "$FAILURE_ALARM" ] || return 1 + if ! fm_lock_try_acquire "$OWNER_LOCK"; then + pid=$(cat "$OWNER_LOCK/pid" 2>/dev/null || true) + role=$(fm_lock_role "$OWNER_LOCK" 2>/dev/null || true) + if fm_pid_alive "$pid" && [ "$role" = autoarm ]; then + return 2 + fi + return 1 + fi + if ! fm_lock_set_role "$OWNER_LOCK" terminal-check; then + fm_lock_release "$OWNER_LOCK" + return 1 + fi + if ! fm_lock_try_acquire "$BUDGET_LOCK"; then + fm_lock_release "$OWNER_LOCK" + return 1 + fi + old_session=$(sed -n '1s/^session=//p' "$BUDGET_FILE" 2>/dev/null || true) + old_count=$(sed -n '2s/^count=//p' "$BUDGET_FILE" 2>/dev/null || true) + case "$old_count" in + ''|*[!0-9]*) old_count=0 ;; + esac + role=$(fm_lock_role "$OWNER_LOCK" 2>/dev/null || true) + if [ "$role" != terminal-check ] || [ "$old_session" != "$SESSION_ID" ] \ + || [ "$old_count" -le "$BLOCK_BUDGET" ] || ! failure_episode_verified \ + || [ -e "$FAILURE_ALARM" ]; then + fm_lock_release "$BUDGET_LOCK" + fm_lock_release "$OWNER_LOCK" + return 1 + fi + if fm_watcher_healthy "$STATE" "$WATCH" "$GRACE" "$FM_HOME"; then + if ! fm_failure_episode_reset "$STATE" held; then + fm_lock_release "$BUDGET_LOCK" + fm_lock_release "$OWNER_LOCK" + return 1 + fi + fm_lock_release "$BUDGET_LOCK" + fm_lock_release "$OWNER_LOCK" + return 2 + fi + if ! (set -C; : > "$FAILURE_ALARM") 2>/dev/null; then + fm_lock_release "$BUDGET_LOCK" + fm_lock_release "$OWNER_LOCK" + return 1 + fi + fm_lock_release "$BUDGET_LOCK" + fm_lock_release "$OWNER_LOCK" + return 0 +} + +failure_episode_verified() { + local outcome + [ ! -e "$STATE/.afk" ] || return 1 + [ -e "$FAILURE_NOTICE" ] || return 1 + outcome=$(sed -n 's/^.*outcome=\([a-z][a-z-]*\) .*$/\1/p' "$STATE/.claude-autoarm-epoch" 2>/dev/null || true) + case "$outcome" in + failed|failed-suppressed) return 0 ;; + *) return 1 ;; + esac +} + i=0 while [ "$i" -lt $((SYNC_WAIT_MS / 100)) ]; do if autoarm_owns_recovery; then - budget_reset + if fm_watcher_healthy "$STATE" "$WATCH" "$GRACE" "$FM_HOME"; then + fm_failure_episode_reset "$STATE" || exit 2 + fi exit 0 fi sleep 0.1 i=$((i + 1)) done if autoarm_owns_recovery; then - budget_reset + if fm_watcher_healthy "$STATE" "$WATCH" "$GRACE" "$FM_HOME"; then + fm_failure_episode_reset "$STATE" || exit 2 + fi exit 0 fi -# The auto-arm genuinely failed to establish: re-block, but never past the -# budget so the session can always end and Claude's 8-block override is never -# approached. -SESSION_ID=$(printf '%s' "$PAYLOAD" | jq -r '.session_id // "unknown"' 2>/dev/null || printf 'unknown') -COUNT=0 -if [ -f "$BUDGET_FILE" ]; then - old_session=$(sed -n '1s/^session=//p' "$BUDGET_FILE" 2>/dev/null || true) - old_count=$(sed -n '2s/^count=//p' "$BUDGET_FILE" 2>/dev/null || true) - case "$old_count" in - ''|*[!0-9]*) old_count=0 ;; - esac - [ "$old_session" = "$SESSION_ID" ] && COUNT=$old_count -fi -COUNT=$((COUNT + 1)) -if [ "$COUNT" -gt "$BLOCK_BUDGET" ]; then - budget_reset +# The auto-arm genuinely failed to establish: consume the bounded re-block +# budget before considering the verified one-time attended fail-open. +budget_account_current_epoch || block_stop +terminal_fail_open +terminal_status=$? +if [ "$terminal_status" -eq 0 ]; then if [ "$FM_SUP_IN_FLIGHT" -gt 0 ]; then NEED_DESC="$FM_SUP_IN_FLIGHT task(s) in flight" elif [ "$FM_SUP_SOURCES" -gt 0 ]; then @@ -229,8 +369,8 @@ if [ "$COUNT" -gt "$BLOCK_BUDGET" ]; then else NEED_DESC="X-mode relay polling active" fi - printf '{"systemMessage":"firstmate turn-end guard: %s with no live watcher and no Stop auto-arm claim; block budget exhausted, allowing this stop. Repair supervision (bin/fm-watch-arm.sh as a Claude Code background task) or investigate why bin/fm-claude-stop-autoarm.sh is not claiming this home."}\n' "$NEED_DESC" + printf '{"systemMessage":"FIRSTMATE SUPERVISION IS GENUINELY DOWN: %s, the Stop-owned auto-arm exhausted its bounded retries and one failure notice, no watcher or automatic continuation exists, and the block budget is exhausted. Keep this session attended and diagnose the automatic Stop-hook and watcher startup before relying on unattended supervision."}\n' "$NEED_DESC" exit 0 fi -printf 'session=%s\ncount=%s\n' "$SESSION_ID" "$COUNT" > "$BUDGET_FILE" 2>/dev/null || true +[ "$terminal_status" -eq 2 ] && exit 0 block_stop diff --git a/bin/fm-wake-lib.sh b/bin/fm-wake-lib.sh index 8c285963934..0aeac114149 100755 --- a/bin/fm-wake-lib.sh +++ b/bin/fm-wake-lib.sh @@ -121,10 +121,29 @@ fm_lock_clean_known_files() { "$lockdir/pid" \ "$lockdir/fm-home" \ "$lockdir/pid-identity" \ + "$lockdir/role" \ "$lockdir/watcher-path" \ 2>/dev/null || true } +fm_lock_set_role() { + local lockdir=$1 role=$2 current pid back + case "$role" in + autoarm|terminal-check) : ;; + *) return 1 ;; + esac + current=${BASHPID:-$$} + pid=$(cat "$lockdir/pid" 2>/dev/null || true) + [ "$pid" = "$current" ] || return 1 + printf '%s\n' "$role" > "$lockdir/role" 2>/dev/null || return 1 + back=$(cat "$lockdir/role" 2>/dev/null || true) + [ "$back" = "$role" ] +} + +fm_lock_role() { + cat "$1/role" 2>/dev/null +} + fm_lock_abs_path() { local path=$1 dir base dir=$(dirname "$path") @@ -383,6 +402,43 @@ fm_lock_release() { rmdir "$lockdir" 2>/dev/null || true } +fm_failure_episode_reset() { + local state=$1 mode=${2:-acquire} lock current pid acquired=0 path + lock="$state/.turnend-claude-blocks.lock" + case "$mode" in + acquire) + fm_lock_try_acquire "$lock" || return 1 + acquired=1 + ;; + held) + current=${BASHPID:-$$} + pid=$(cat "$lock/pid" 2>/dev/null || true) + [ "$pid" = "$current" ] || return 1 + ;; + *) return 1 ;; + esac + for path in \ + "$state/.turnend-claude-blocks" \ + "$state/.claude-autoarm-failure-notified" \ + "$state/.claude-autoarm-failure-alarmed" + do + if [ -d "$path" ] && [ ! -L "$path" ]; then + [ "$acquired" -eq 0 ] || fm_lock_release "$lock" + return 1 + fi + done + if ! rm -f \ + "$state/.turnend-claude-blocks" \ + "$state/.claude-autoarm-failure-notified" \ + "$state/.claude-autoarm-failure-alarmed" \ + 2>/dev/null; then + [ "$acquired" -eq 0 ] || fm_lock_release "$lock" + return 1 + fi + [ "$acquired" -eq 0 ] || fm_lock_release "$lock" + return 0 +} + fm_wake_clean_field() { LC_ALL=C tr '\t\r\n' ' ' } diff --git a/docs/architecture.md b/docs/architecture.md index c94a95dd4f3..9507d41b777 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -60,14 +60,15 @@ That block owns the live wait shape for the running primary harness: Claude's St [`watcher-continuity.md`](watcher-continuity.md#arm-layer-cycle-contract) owns the arm layer's successor, terminal-delivery, and typed clean-close failure contract. The arm layer records one bounded lifecycle row per observed cycle in `state/.watch-cycle-exits.log`; `state/.watch-triage.log` remains exclusively the absorbed-wake debug log. Pi and OpenCode verify session-lock ownership and launch one singleton successor from their child-close handlers before delivering an actionable wake prompt, with bounded exponential retry for failed restoration. -Claude's `bin/fm-claude-stop-autoarm.sh` hook fires on every Stop and, when the home is eligible and still needs supervision, claims one home-scoped cycle, foregrounds the arm wrapper, and translates an actionable close or typed failure into one exit-2 rewake. +Claude's `bin/fm-claude-stop-autoarm.sh` hook fires on every Stop and, when the home is eligible and still needs supervision, claims one home-scoped cycle, foregrounds the arm wrapper, and translates actionable closes into exit-2 rewakes. +It suppresses failed-looking closes when the same identity-matched watcher is healthy, retries genuine failures within a bound, and coordinates exhausted failure episodes with the Claude turn-end guard as documented in [`turnend-guard.md`](turnend-guard.md). [`watcher-continuity.md`](watcher-continuity.md) owns Claude's residual active-turn coverage and watcher-status command-gating boundary. The existing turn-end guard remains the final backstop for all five harness-engine protocols, with pi-signed sharing Pi's protocol and the `--claude` mode cooperating with the auto-arm claim. Its `--restart` mode signals only the watcher recorded in the current home's `state/.watch.lock`, so restarting one home cannot kill sibling secondmate watchers. -A pull-based guard (`bin/fm-guard.sh`) warns through supervision tool output if the primary checkout is tangled, or if tasks are in flight and that watcher stops running or queued wakes are waiting to be drained. +A pull-based guard (`bin/fm-guard.sh`) warns through supervision tool output if the primary checkout is tangled, if work, process-event sources, or X-mode relay polling needs supervision without a healthy identity-matched watcher, or if queued wakes are waiting to be drained. The drain script calls that guard after emptying the queue, which avoids repeating the queued-wakes warning for records it just consumed while still warning on stale watcher liveness. It leads with a prominent bordered tangle banner, while `bin/fm-guard.sh` owns the stale-watcher banner/reminder policy so repeated guarded commands stay noisy without reprinting the full watcher-down banner in the same episode. -On every verified primary harness, tracked hook integration gives the primary session a push-based backstop: when work is in flight and no identity-matched watcher lock with a fresh beacon is live, direct Stop hooks block and passive turn-end hooks force one bounded follow-up. +On every verified primary harness, tracked hook integration gives the primary session a push-based backstop: when work, a process-event source, or X-mode relay polling needs supervision and no identity-matched watcher lock with a fresh beacon is live, direct Stop hooks block and passive turn-end hooks force one bounded follow-up. The guard covers the main primary and genuinely marked secondmate homes, exempts child crewmate/scout worktrees, is loop-safe per harness, and is documented in [turnend-guard.md](turnend-guard.md). A presence-gated sub-supervisor (`bin/fm-supervise-daemon.sh`) extends this for walk-away supervision: the `/afk` skill starts it through the tracked foreground helper `bin/fm-afk-start.sh`, after which the watcher reverts to daemon-managed one-shot mode and the daemon self-handles routine wakes in bash. diff --git a/docs/configuration.md b/docs/configuration.md index cc6033a0b07..ba50bca1db8 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -505,9 +505,10 @@ FMX_FOLLOWUP_MAX_COUNT=3 # local cap on X-mode completion follow-ups per linke FM_PF_RETRY_BACKOFF_SECS=900 # seconds before the next attempt after a retryable promised-public-reply delivery error FM_LOCK_STALE_AFTER=2 # seconds before dead-pid lock records can be reclaimed; mid-acquire locks keep at least 2s grace FM_GUARD_GRACE=300 # seconds before guard warnings, arm health checks, and the primary turn-end guard treat a watcher beacon as stale -FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=800 # milliseconds the --claude turn-end guard waits for the Stop auto-arm's claim, health, or fresh rewake epoch before re-blocking -FM_CLAUDE_AUTOARM_EPOCH_FRESH=15 # seconds a recorded auto-arm rewake outcome counts as this event epoch's owned recovery -FM_CLAUDE_TURNEND_BLOCK_BUDGET=3 # consecutive --claude guard re-blocks before a degraded allow; safely below Claude Code's 8-block override +FM_CLAUDE_AUTOARM_ATTEMPTS=2 # bounded Stop-owned arm attempts per Claude auto-arm cycle; accepted values are 1, 2, or 3 +FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=800 # milliseconds the --claude turn-end guard waits for watcher health, a role-verified Stop auto-arm claim, or a fresh epoch before deciding recovery ownership or failure progression +FM_CLAUDE_AUTOARM_EPOCH_FRESH=15 # seconds a recorded auto-arm outcome remains eligible for the current event epoch's recovery or failure decision +FM_CLAUDE_TURNEND_BLOCK_BUDGET=3 # consecutive --claude guard re-blocks before the verified one-time attended fail-open; safely below Claude Code's 8-block override FM_ARM_CONFIRM_TIMEOUT=10 # seconds fm-watch-arm waits to confirm a fresh watcher before reporting FAILED; default 30 on Git Bash/MSYS FM_ARM_ATTACH_POLL=0.5 # seconds between checks while fm-watch-arm is attached to an existing healthy watcher cycle FM_OPENCODE_ARM_READY_TIMEOUT_MS=12000 # milliseconds the OpenCode primary watcher plugin waits for an arm attempt to report started, healthy, wake, or failure; default 35000 on Windows to stay above the MSYS confirm budget diff --git a/docs/supervision-protocols/claude.md b/docs/supervision-protocols/claude.md index c9913553102..2b32be24d78 100644 --- a/docs/supervision-protocols/claude.md +++ b/docs/supervision-protocols/claude.md @@ -8,18 +8,17 @@ When this session owns supervision and away mode is not active: 3. On a `Stop hook feedback` wake (`signal:`, `stale:`, `check:`, or `heartbeat`), run `bin/fm-wake-drain.sh` first and handle the wake. Do not run `bin/fm-watch-arm.sh` after an ordinary wake; the next turn end re-arms automatically when supervision is still needed. Do not invent a wake from an attach-status line alone; drain and act only on real wake records or a real watcher reason line. -4. On a `Stop hook feedback` watcher-failure wake (`watcher: FAILED ...`), treat it as an alarm: drain, then repair supervision before ending the turn. -5. Manual arm is recovery only. - When a repair is genuinely needed - the Stop hook did not claim this home, or a forced restart is required - run `bin/fm-watch-arm.sh` (or `bin/fm-watch-arm.sh --restart`) as its own Claude Code background task, never bundled with other commands, never with shell `&`. - Source `__FM_X_MODE_ENV__` first when X mode is active. - A shell `&`, a truncating pipe, or bundling is denied automatically by the PreToolUse seatbelt (`bin/fm-arm-pretool-check.sh`) registered in `.claude/settings.json`. -6. Treat `watcher: started ...` and `watcher: attached ...` inside arm output as proof that one live cycle exists. +4. On the one `Stop hook feedback` automatic-mechanism failure notice (`firstmate watcher auto-arm FAILED ...`), drain, inspect the automatic mechanism failure, and do not turn the notice into a repeating manual-arm loop. +5. If the Stop hook does not claim the home or reports an exhausted failure, inspect its registration and watcher startup path before ending blind. + Keep the Stop-owned automatic mechanism as the only Claude arm owner. +6. Treat `watcher: started ...` and `watcher: attached ...` inside automatic arm output as proof that one live cycle exists. On attach, the arm follows verified identity-matched successors instead of exiting when the first cycle ends. 7. The durable wake queue preserves actionable events between a rewake and the next Stop-launched arm, while the bounded turn-end guard prevents a blind Stop when recovery did not start. No PreToolUse hook denies fleet commands based on watcher status. [`watcher-continuity.md`](../watcher-continuity.md) owns the exact session-lock recovery boundary. 8. The turn-end guard (`bin/fm-turnend-guard.sh --claude`) remains the final backstop. - It allows the stop when a watcher is healthy, when the auto-arm already owns recovery for this event epoch, or when a fresh rewake is recorded; it re-blocks only when none of those materialize, within a bounded budget. + It uses the same live-watcher and fresh-beacon predicate as the pull guard. + It allows the stop when a watcher is healthy or the role-verified auto-arm owns recovery, while fresh failure epochs advance the bounded one-time attended fail-open progression described in [`turnend-guard.md`](../turnend-guard.md). 9. Waiting on the hook-owned cycle is silent: do not send idle progress while the watcher is parked. The watcher itself remains `bin/fm-watch.sh`, and `bin/fm-watch-arm.sh` remains the verified arm wrapper that the Stop hook foregrounds. diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index 8ee750de397..63d13be70f4 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -13,7 +13,8 @@ Do not infer this guard's scope, loop safety, or compatibility tradeoffs for tho `bin/fm-guard.sh` is a pull-based warning that runs only when another supervision command invokes it. The turn-end guard closes the remaining gap at the primary's own turn boundary. -When work is in flight and no identity-matched watcher has a fresh beacon, the harness integration must either block the turn end or force one bounded follow-up that uses the recovery instruction from the emitted session-start protocol. +When work, a process-event source, or X-mode relay polling needs supervision and no identity-matched watcher has a fresh beacon, the harness integration must either block the turn end or force one bounded follow-up that uses the recovery instruction from the emitted session-start protocol. +Both guards require the same live lock, process identity, home/path binding, and fresh-beacon predicate. The guard remains a backstop; [`watcher-continuity.md`](watcher-continuity.md) owns normal continuity. ## Shared predicate @@ -26,9 +27,11 @@ That check keeps crewmate and scout linked worktrees inert because their git dir It also requires `AGENTS.md`, `bin/`, and the effective state directory. For an in-scope primary, the guard counts in-flight work from `state/*.meta`. -The default cross-harness mode exits silently with no work in flight. -Claude's `--claude` mode also treats `state/x-watch.check.sh` as supervision need, so X-mode relay polling remains guarded without an in-flight task. +Registered `state/procevent/*.source` records also require supervision even though they have no task metadata. +The default cross-harness mode exits silently with no supervision need. +Every mode treats `state/x-watch.check.sh` as supervision need, so X-mode relay polling remains guarded without an in-flight task. Otherwise it calls `fm_watcher_healthy [grace-seconds] [home]` from `bin/fm-wake-lib.sh`, the same identity-matched lock and fresh-beacon check used by `bin/fm-watch-arm.sh`. +`bin/fm-guard.sh` uses that same check rather than treating the status helper's fresh-beacon field as sufficient. A stale beacon blocks even when a watcher pid is live. A fresh leftover beacon blocks when the lock is missing, dead, or identity-mismatched. @@ -51,9 +54,19 @@ In the default Codex mode, a true value lets the second stop finish after one fo Claude runs the guard with `--claude`, which ignores `stop_hook_active` and cooperates with the Stop-owned auto-arm. Claude Code sets `stop_hook_active=true` on every stop after any stop-hook continuation, including `asyncRewake` rewakes, which re-opened the 2026-07-21 blind window under the default one-shot behavior. -The Claude mode waits up to `FM_CLAUDE_AUTOARM_SYNC_WAIT_MS` (default 800 milliseconds) and allows the stop when the watcher is healthy, `state/.claude-autoarm.lock` has a live owner, or `state/.claude-autoarm-epoch` contains a fresh rewake outcome. -When none of those proofs appears, it re-blocks up to `FM_CLAUDE_TURNEND_BLOCK_BUDGET` times (default 3, below Claude's 8-block override), then allows degraded with a visible `systemMessage`. -Any allow resets the budget. +The Claude mode waits up to `FM_CLAUDE_AUTOARM_SYNC_WAIT_MS` (default 800 milliseconds) and allows the stop when the watcher is healthy, `state/.claude-autoarm.lock` has a live `autoarm` role owner whose eventual failure must exit 2, or `state/.claude-autoarm-epoch` contains a fresh actionable rewake owned by this event epoch. +Fresh `failed` and `failed-suppressed` outcomes enter or advance the failure progression instead of acting as unconditional recovery proof. +The auto-arm itself rechecks the healthy watcher predicate and retries a bounded number of times before reporting a genuine failure. +The first fresh exhausted-failure epoch preserves its handoff without consuming a blocked-stop count, while later fresh failed epochs advance the same monotonic progression instead of resetting it. +When none of those proofs appears, it re-blocks up to `FM_CLAUDE_TURNEND_BLOCK_BUDGET` times (default 3, below Claude's 8-block override). +In Claude mode, positive watcher recovery clears the block budget, failure notice, and attended alarm together under the existing budget lock before either hook reports ordinary recovery. +The one loud attended fail-open is available only when the auto-arm has recorded an exhausted failure, its one notice is already consumed, the block budget is exhausted, and a final check finds neither a healthy watcher nor an automatic continuation. +Each epoch identity is accounted at most once under the budget lock. +Whenever both coordination locks are needed, positive auto-arm recovery and the terminal check acquire the auto-arm owner lock before the budget lock. +After that alarm, the Stop auto-arm suppresses further exit-2 continuations until positive watcher recovery, so the final fail-open remains reachable. +The alarm cannot repeat during that failure episode, and a later unhealthy stop blocks again. +A positively verified healthy watcher clears the failure notice, alarm, and block budget for a future independent episode. +A Claude failure notice describes the automatic mechanism as broken and does not direct a routine manual background arm. OpenCode, Pi, and pi-signed expose passive callbacks for this purpose. Their adapters fail open at the hook boundary to protect the user session but schedule one bounded follow-up when the predicate blocks. @@ -90,7 +103,8 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa ## Regression coverage -`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the cooperative `--claude` claim wait, epoch allow, re-block budget, Pi logical-run latching, missing-`jq` behavior, all five primary registrations, Grok native and legacy selection, typed field precedence, malformed input, and exactly-one-path safety. +`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the live-lock and fresh-beacon guard predicate, the cooperative `--claude` claim wait, monotonic failed-epoch progression, bounded attended fail-open, post-alarm continuation suppression, positive recovery reset, Pi logical-run latching, missing-`jq` behavior, all five primary registrations, Grok native and legacy selection, typed field precedence, malformed input, and exactly-one-path safety. +`tests/fm-guard-stale-banner.test.sh` covers the matching pull-guard predicate, including the fresh-leftover-beacon negative control. `tests/fm-kimi-harness.test.sh` covers the separate Kimi crew hook's format preservation, idempotence, refusal cases, token guard, spawn registration, and teardown cleanup. `tests/fm-supervision-instructions.test.sh` covers recovery-line ownership and pi-signed's identity-preserving reuse of Pi's protocol. `FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh` is the opt-in isolated Pi path. diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md index 00b13c1dd4a..a9ed234fb58 100644 --- a/docs/verification/supervision.md +++ b/docs/verification/supervision.md @@ -155,6 +155,46 @@ FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh FM_GROK_STOP_LIVE_E2E=1 FM_GROK_NATIVE_BIN="$native_grok" FM_GROK_LEGACY_BIN="$pre_native_grok" tests/fm-grok-stop-live-e2e.test.sh ``` +The Claude auto-arm false-failure, guard-predicate, and monotonic bounded fail-open correction was verified on 2026-08-02 with the installed ShellCheck 0.11.0 and isolated behavior suites. + +```sh +bin/fm-lint.sh +bin/fm-doc-audience-check.sh +bin/fm-test-run.sh tests/fm-claude-stop-autoarm.test.sh tests/fm-guard-stale-banner.test.sh tests/fm-turnend-guard.test.sh tests/fm-supervision-instructions.test.sh +``` + +Observed output: + +```text +fm-lint.sh: ShellCheck 0.11.0 (pinned 0.11.0) +fm-doc-audience-check: ok surfaces=61 local_links=174 +FM_TEST_SUMMARY total=4 failed=0 skipped_gate=0 duration_ms=102585 +``` + +The broader relevant regression pass was rerun on 2026-08-02 without live-home or daemon mutation. + +```sh +bin/fm-test-run.sh tests/fm-watch-triage.test.sh tests/fm-watcher-lock.test.sh tests/fm-afk-inject-e2e.test.sh tests/fm-afk-return.test.sh tests/fm-x-mode.test.sh tests/fm-backend.test.sh tests/fm-backend-tmux-smoke.test.sh tests/fm-secondmate-safety.test.sh +``` + +Observed output: + +```text +FM_TEST_SUMMARY total=8 failed=0 skipped_gate=0 duration_ms=617507 +``` + +The actionable-close ordering correction was reverified on 2026-08-02 against an identity-matched live successor. + +```sh +tests/fm-claude-stop-autoarm.test.sh >/dev/null && echo "fm-claude-stop-autoarm: ok" +``` + +Observed output: + +```text +fm-claude-stop-autoarm: ok +``` + ## Watcher continuity The cross-harness evidence combines the 2026-07-17 live pass with Claude's replacement Stop-owned path revalidated on 2026-07-24, all against isolated project and home state. diff --git a/docs/watcher-continuity.md b/docs/watcher-continuity.md index 952f7cca88d..2ae9a6b17bb 100644 --- a/docs/watcher-continuity.md +++ b/docs/watcher-continuity.md @@ -13,7 +13,11 @@ Claude's `.claude/settings.json` Stop `asyncRewake` hook (`bin/fm-claude-stop-au The hook fires on every Stop, and an eligible primary with supervision need admits one home-scoped owner that foregrounds `bin/fm-watch-arm.sh` inside the hook-owned process tree. A numeric session-lock owner that fails the shared `fm_harness_pid_alive` predicate is reclaimed through `bin/fm-lock.sh` before auto-arm state changes, while a live owner, absent lock, or malformed lock keeps the competing hook inert. The stale-owner claim occurs only after the existing AFK and supervision-need gates pass. -While supervision is still needed and away mode remains inactive, an actionable close or typed failure wakes the idle session through exit 2. +After each non-actionable arm close, the hook rechecks the identity-matched watcher lock and fresh beacon before retrying a bounded number of times. +A cycle-end failure is benign when that live-watcher predicate is true, and the hook suppresses the arm output and continues silently. +Only an exhausted failure with no verified watcher emits one last-resort notice for the continuous failure episode; later consecutive Stop cycles exit 2 to guarantee another Stop-owned retry without repeating the notice until the turn-end guard consumes the attended fail-open. +The Claude turn-end guard owns the monotonic failure progression, one-time attended fail-open, post-alarm continuation suppression, and positive recovery reset described in [`turnend-guard.md`](turnend-guard.md#harness-integrations). +While supervision is still needed and away mode remains inactive, an actionable close wakes the idle session through exit 2. ## Actionable wake ordering @@ -25,9 +29,10 @@ After the configured retry bound is exhausted, it delivers the original wake wit This is deliberate Option B ordering: the fleet is protected before the model handles the wake whenever restoration succeeds, but the model is never left blind when it does not. Claude's Stop hook starts the successor arm at the next Stop after the handling turn, rather than before notification as Pi and OpenCode do. -The durable wake queue preserves actionable events during the residual active-turn window, and the unchanged bounded turn-end guard enforces recovery at Stop when no watcher or auto-arm claim is present. -No PreToolUse hook denies fleet commands based on watcher status. +The durable wake queue preserves actionable events during the residual active-turn window, and the bounded turn-end guard enforces recovery at Stop when no watcher or auto-arm claim is present. The model no longer re-arms after ordinary wakes. +No PreToolUse hook denies fleet commands based on watcher status. +A genuine auto-arm failure describes the automatic mechanism as broken and never directs a routine manual background arm. Terminal arm-output classification (`started`, `attached`, or `FAILED`) remains defense in depth for the manual recovery path. Codex retains its bounded foreground checkpoint protocol. Grok retains its tracked background-task notification protocol. @@ -59,9 +64,9 @@ Only the watcher process touches `state/.last-watcher-beat`; no helper process c The same suite covers ordinary same-process session replacement for `/new`, `/resume`, and `/fork`, same-instance shutdown-plus-start, stale prior-generation callbacks, repeated transitions with exactly one live cycle, disappearance of the shutting-down refusal after a valid replacement activates, and terminal quit still refusing late rearm. `tests/fm-watcher-lock.test.sh` covers verified-successor attach, the typed self-eviction failure, bounded and successor-linked lifecycle rows, and a SIGSTOP counterfactual that distinguishes a live PID from a stale beacon before classifying termination. `tests/fm-subagent-pretool-check.test.sh` proves Claude retains only the non-status Bash seatbelts. -`tests/fm-claude-stop-autoarm.test.sh` covers the auto-arm's scope, stale and live session owners, unchanged AFK and need boundaries, single-flight, and exit-2 translation. +`tests/fm-claude-stop-autoarm.test.sh` covers the auto-arm's scope, stale and live session owners, unchanged AFK and need boundaries, single-flight, bounded failure retries, benign live-watcher cycle ends, one-notice failure episodes, and exit-2 translation. `FM_CLAUDE_LIVE_E2E=1 tests/fm-claude-stop-autoarm-live-e2e.test.sh` starts with the reproduced stale-lock state, runs session start first, completes two tokenless cycles, and checks the competing-live-owner negative control. -`tests/fm-turnend-guard.test.sh` covers the cooperative `--claude` guard. +`tests/fm-turnend-guard.test.sh` covers the cooperative `--claude` guard, including monotonic failed-epoch progression, the integrated bounded fail-open, post-alarm continuation suppression, and positive recovery reset. ## Active limits and verification diff --git a/tests/fm-claude-stop-autoarm.test.sh b/tests/fm-claude-stop-autoarm.test.sh index 6be8bc15333..f0901667913 100755 --- a/tests/fm-claude-stop-autoarm.test.sh +++ b/tests/fm-claude-stop-autoarm.test.sh @@ -104,6 +104,14 @@ SH echo "$$" >> "$FM_HOME/state/arm-ran" printf 'watcher: attached pid=%s (beacon 2s)\n' "$$" exit 0 +SH + ;; + benign-live) + cat > "$dir/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +echo "$$" >> "$FM_HOME/state/arm-ran" +printf 'watcher: FAILED - cycle ended without an actionable reason\n' +exit 1 SH ;; slow-actionable) @@ -145,7 +153,23 @@ SH } epoch_outcome() { - sed -n 's/^.*outcome=\([a-z][a-z]*\) .*$/\1/p' "$1/state/.claude-autoarm-epoch" 2>/dev/null || true + sed -n 's/^.*outcome=\([a-z][a-z-]*\) .*$/\1/p' "$1/state/.claude-autoarm-epoch" 2>/dev/null || true +} + +watcher_identity() { + local dir=$1 pid=$2 + FM_STATE_OVERRIDE="$dir/state" bash -c '. "$1"; fm_pid_identity "$2"' _ "$dir/bin/fm-wake-lib.sh" "$pid" +} + +record_watcher_lock() { + local dir=$1 pid=$2 identity=$3 root bin_dir + root=$dir + bin_dir=$(cd "$dir/bin" && pwd) + mkdir -p "$dir/state/.watch.lock" + printf '%s\n' "$pid" > "$dir/state/.watch.lock/pid" + printf '%s\n' "$root" > "$dir/state/.watch.lock/fm-home" + printf '%s\n' "$bin_dir/fm-watch.sh" > "$dir/state/.watch.lock/watcher-path" + printf '%s\n' "$identity" > "$dir/state/.watch.lock/pid-identity" } # --- registration contract ---------------------------------------------------- @@ -224,10 +248,14 @@ test_inert_when_afk() { dir=$(make_primary_dir "$TMP_ROOT/afk") : > "$dir/state/task.meta" : > "$dir/state/.afk" + : > "$dir/state/.claude-autoarm-failure-notified" + : > "$dir/state/.claude-autoarm-failure-alarmed" write_arm_fixture "$dir" actionable out=$(run_autoarm "$dir" 2>/dev/null); status=$? expect_code 0 "$status" "hook must never arm or rewake while away mode owns triage" [ ! -e "$dir/state/arm-ran" ] || fail "hook armed while state/.afk existed" + assert_present "$dir/state/.claude-autoarm-failure-notified" "AFK without positive recovery reset the failure notice" + assert_present "$dir/state/.claude-autoarm-failure-alarmed" "AFK without positive recovery reset the attended alarm" pass "auto-arm: inert while AFK owns supervision" } @@ -287,10 +315,14 @@ test_resolves_outermost_claude_pid_in_nested_bgspare_chain() { test_inert_when_fleet_idle() { local dir out status dir=$(make_primary_dir "$TMP_ROOT/idle") + : > "$dir/state/.claude-autoarm-failure-notified" + : > "$dir/state/.claude-autoarm-failure-alarmed" write_arm_fixture "$dir" actionable out=$(run_autoarm "$dir" 2>/dev/null); status=$? expect_code 0 "$status" "hook must exit 0 in an idle home with no X-mode poll" [ ! -e "$dir/state/arm-ran" ] || fail "hook armed an idle home" + assert_present "$dir/state/.claude-autoarm-failure-notified" "idle state without positive recovery reset the failure notice" + assert_present "$dir/state/.claude-autoarm-failure-alarmed" "idle state without positive recovery reset the attended alarm" pass "auto-arm: inert with nothing in flight and no X-mode need" } @@ -313,6 +345,36 @@ test_actionable_close_rewakes_with_reason() { pass "auto-arm: actionable close translates to exactly one exit-2 rewake with reason" } +test_actionable_close_with_live_successor_rewakes_once() { + local dir out out2 status status2 pid identity + dir=$(make_primary_dir "$TMP_ROOT/actionable-live-successor") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" actionable + sleep 60 & + pid=$! + identity=$(watcher_identity "$dir" "$pid") || fail "could not identify live successor for actionable close" + record_watcher_lock "$dir" "$pid" "$identity" + touch "$dir/state/.last-watcher-beat" + + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + write_arm_fixture "$dir" benign-live + out2=$(run_autoarm "$dir" 2>/dev/null); status2=$? + + expect_code 2 "$status" "an actionable close must rewake when a live successor already exists" + expect_code 0 "$status2" "a repeated non-actionable close with the live successor must stay quiet" + [ "$(printf '%s\n' "$out" | grep -c '^firstmate watcher wake')" -eq 1 ] \ + || fail "actionable close with a live successor did not emit exactly one wake banner: $out" + [ "$(printf '%s\n' "$out" | grep -c '^stale: fixture-win actionable')" -eq 1 ] \ + || fail "actionable close with a live successor did not surface its reason exactly once: $out" + [ -z "$out2" ] || fail "repeated hook duplicated the delivered actionable result: $out2" + kill -0 "$pid" 2>/dev/null || fail "actionable delivery stopped or replaced the live successor" + [ "$(epoch_outcome "$dir")" = clean ] || fail "the later benign close must record outcome=clean" + + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + pass "auto-arm: actionable close survives a healthy successor without duplicate delivery" +} + test_failed_close_rewakes_with_failure_banner() { local dir out status dir=$(make_primary_dir "$TMP_ROOT/failed") @@ -320,23 +382,120 @@ test_failed_close_rewakes_with_failure_banner() { write_arm_fixture "$dir" failed out=$(run_autoarm "$dir" 2>/dev/null); status=$? expect_code 2 "$status" "a typed watcher failure must rewake as an alarm" - assert_contains "$out" "watcher cycle FAILED" "failure rewake must carry the failure banner" + assert_contains "$out" "automatic supervision mechanism is broken" "failure rewake must describe the automatic mechanism failure" assert_contains "$out" "watcher: FAILED" "failure rewake must carry the arm's typed failure" - assert_contains "$out" "repair supervision" "failure rewake must direct the manual repair" - [ "$(epoch_outcome "$dir")" = rewake ] || fail "epoch must record outcome=rewake, got: $(epoch_outcome "$dir")" - pass "auto-arm: watcher: FAILED translates to an exit-2 alarm rewake" + assert_not_contains "$out" "bin/fm-watch-arm.sh" "failure rewake must not create a manual arm loop" + [ "$(epoch_outcome "$dir")" = failed ] || fail "epoch must record outcome=failed, got: $(epoch_outcome "$dir")" + [ "$(wc -l < "$dir/state/arm-ran" | tr -d ' ')" -eq 2 ] || fail "failure must exhaust exactly two bounded arm attempts" + pass "auto-arm: bounded failure verification emits one automatic-mechanism alarm" } -test_clean_close_exits_silently() { +test_failed_cycles_notify_once_and_keep_retrying() { + local dir out1 out2 status1 status2 + dir=$(make_primary_dir "$TMP_ROOT/failed-dedup") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" failed + out1=$(run_autoarm "$dir" 2>/dev/null); status1=$? + out2=$(run_autoarm "$dir" 2>/dev/null); status2=$? + expect_code 2 "$status1" "the first exhausted failure must notify" + expect_code 2 "$status2" "a consecutive exhausted failure must force another Stop-owned retry" + [ -n "$out1" ] || fail "the first exhausted failure did not notify" + [ -z "$out2" ] || fail "consecutive exhausted failure repeated an operator notice: $out2" + [ "$(wc -l < "$dir/state/arm-ran" | tr -d ' ')" -eq 4 ] || fail "each cycle must retain bounded automatic retries" + assert_present "$dir/state/.claude-autoarm-failure-notified" "failure episode marker was not recorded" + [ "$(epoch_outcome "$dir")" = failed-suppressed ] || fail "second failure must record failed-suppressed" + pass "auto-arm: consecutive failures keep Stop-owned retry without repeating notice" +} + +test_unverified_clean_close_exhausts_retries() { local dir out status dir=$(make_primary_dir "$TMP_ROOT/clean") : > "$dir/state/task.meta" write_arm_fixture "$dir" clean out=$(run_autoarm "$dir" 2>/dev/null); status=$? - expect_code 0 "$status" "a clean arm close with no actionable reason must not rewake" - [ -z "$out" ] || fail "clean close produced output: $out" - [ "$(epoch_outcome "$dir")" = clean ] || fail "epoch must record outcome=clean, got: $(epoch_outcome "$dir")" - pass "auto-arm: clean close exits silently with a clean epoch" + expect_code 2 "$status" "a non-actionable close without a healthy watcher must fail closed" + assert_contains "$out" "automatic supervision mechanism is broken" "unverified close must report automatic failure" + [ "$(wc -l < "$dir/state/arm-ran" | tr -d ' ')" -eq 2 ] || fail "unverified close must exhaust exactly two bounded attempts" + [ "$(epoch_outcome "$dir")" = failed ] || fail "epoch must record outcome=failed, got: $(epoch_outcome "$dir")" + pass "auto-arm: unverified clean close exhausts retries and fails closed" +} + +test_post_alarm_actionable_close_is_suppressed() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/post-alarm-actionable") + : > "$dir/state/task.meta" + : > "$dir/state/.claude-autoarm-failure-notified" + : > "$dir/state/.claude-autoarm-failure-alarmed" + write_arm_fixture "$dir" actionable + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 0 "$status" "an actionable result after attended fail-open must not continue" + [ -z "$out" ] || fail "post-alarm actionable result produced continuation output: $out" + assert_present "$dir/state/.claude-autoarm-failure-notified" "post-alarm actionable result cleared the failure notice" + assert_present "$dir/state/.claude-autoarm-failure-alarmed" "post-alarm actionable result cleared the attended alarm" + [ "$(epoch_outcome "$dir")" = failed-suppressed ] || fail "post-alarm actionable result must record failed-suppressed" + pass "auto-arm: post-alarm actionable outcomes cannot continue or reset failure state" +} + +test_benign_cycle_end_with_live_watcher_is_silent() { + local dir out out2 status status2 pid identity + dir=$(make_primary_dir "$TMP_ROOT/benign-live") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" benign-live + sleep 60 & + pid=$! + identity=$(watcher_identity "$dir" "$pid") || fail "could not identify live watcher holder for benign close" + record_watcher_lock "$dir" "$pid" "$identity" + touch "$dir/state/.last-watcher-beat" + printf 'session=sess-autoarm\ncount=3\nepoch=9\n' > "$dir/state/.turnend-claude-blocks" + : > "$dir/state/.claude-autoarm-failure-notified" + : > "$dir/state/.claude-autoarm-failure-alarmed" + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + out2=$(run_autoarm "$dir" 2>/dev/null); status2=$? + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + expect_code 0 "$status" "a failed-looking cycle with a live fresh watcher must be benign" + expect_code 0 "$status2" "the next Stop-owned cycle must remain benign with the live watcher" + [ -z "$out" ] || fail "benign live cycle produced an operator notice: $out" + [ -z "$out2" ] || fail "next benign live cycle produced an operator notice: $out2" + [ "$(epoch_outcome "$dir")" = clean ] || fail "benign live cycle must record outcome=clean, got: $(epoch_outcome "$dir")" + [ "$(wc -l < "$dir/state/arm-ran" | tr -d ' ')" -eq 2 ] || fail "the next Stop-owned cycle must run its own bounded arm" + [ ! -e "$dir/state/.turnend-claude-blocks" ] || fail "benign live cycle must clear the prior block budget" + [ ! -e "$dir/state/.claude-autoarm-failure-notified" ] || fail "benign live cycle must not leave a failure-notice marker" + [ ! -e "$dir/state/.claude-autoarm-failure-alarmed" ] || fail "benign live cycle must not leave an attended-alarm marker" + pass "auto-arm: benign cycle end with a live watcher and fresh beacon stays silent across the next cycle" +} + +test_positive_recovery_budget_contention_preserves_episode() { + local dir out status pid identity holder + dir=$(make_primary_dir "$TMP_ROOT/recovery-budget-contention") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" benign-live + sleep 60 & + pid=$! + identity=$(watcher_identity "$dir" "$pid") || fail "could not identify live watcher holder for recovery contention" + record_watcher_lock "$dir" "$pid" "$identity" + touch "$dir/state/.last-watcher-beat" + printf 'session=sess-autoarm\ncount=3\nepoch=9\n' > "$dir/state/.turnend-claude-blocks" + : > "$dir/state/.claude-autoarm-failure-notified" + sleep 60 & + holder=$! + mkdir -p "$dir/state/.turnend-claude-blocks.lock" + printf '%s\n' "$holder" > "$dir/state/.turnend-claude-blocks.lock/pid" + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 2 "$status" "a healthy auto-arm must continue when the episode reset lock is busy" + [ -z "$out" ] || fail "recovery contention produced an operator notice: $out" + [ "$(epoch_outcome "$dir")" = failed-suppressed ] || fail "recovery contention must not record ordinary clean recovery" + assert_present "$dir/state/.turnend-claude-blocks" "recovery contention partially cleared the block budget" + assert_present "$dir/state/.claude-autoarm-failure-notified" "recovery contention partially cleared the failure notice" + kill "$holder" 2>/dev/null || true + wait "$holder" 2>/dev/null || true + out=$(run_autoarm "$dir" 2>/dev/null); status=$? + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + expect_code 0 "$status" "a later healthy auto-arm must complete the episode reset" + assert_absent "$dir/state/.turnend-claude-blocks" "successful retry left the block budget" + assert_absent "$dir/state/.claude-autoarm-failure-notified" "successful retry left the failure notice" + pass "auto-arm: budget contention preserves the episode and forces a reset retry" } test_arms_for_x_mode_poll_need_without_inflight() { @@ -425,8 +584,13 @@ test_stale_lock_recovery_preserves_afk_and_need_gates test_resolves_outermost_claude_pid_in_nested_bgspare_chain test_inert_when_fleet_idle test_actionable_close_rewakes_with_reason +test_actionable_close_with_live_successor_rewakes_once test_failed_close_rewakes_with_failure_banner -test_clean_close_exits_silently +test_failed_cycles_notify_once_and_keep_retrying +test_unverified_clean_close_exhausts_retries +test_post_alarm_actionable_close_is_suppressed +test_benign_cycle_end_with_live_watcher_is_silent +test_positive_recovery_budget_contention_preserves_episode test_arms_for_x_mode_poll_need_without_inflight test_single_flight_admits_exactly_one_owner test_need_vanished_mid_cycle_closes_quietly diff --git a/tests/fm-guard-stale-banner.test.sh b/tests/fm-guard-stale-banner.test.sh index 862eb9c7043..54091035dcb 100755 --- a/tests/fm-guard-stale-banner.test.sh +++ b/tests/fm-guard-stale-banner.test.sh @@ -30,6 +30,17 @@ case_root() { printf '%s/root\n' "$1" } +record_live_watcher() { + local dir=$1 pid=$2 home identity + home=$(case_home "$dir") + identity=$(FM_STATE_OVERRIDE="$home/state" bash -c '. "$1"; fm_pid_identity "$2"' _ "$ROOT/bin/fm-wake-lib.sh" "$pid") || return 1 + mkdir -p "$home/state/.watch.lock" + printf '%s\n' "$pid" > "$home/state/.watch.lock/pid" + printf '%s\n' "$home" > "$home/state/.watch.lock/fm-home" + printf '%s\n' "$ROOT/bin/fm-watch.sh" > "$home/state/.watch.lock/watcher-path" + printf '%s\n' "$identity" > "$home/state/.watch.lock/pid-identity" +} + run_guard_case() { local dir=$1 FM_ROOT_OVERRIDE="$(case_root "$dir")" \ @@ -85,21 +96,48 @@ test_repeated_same_episode_prints_reminder_only() { pass "fm-guard stale banner: repeated same-episode calls print a concise reminder only" } +test_fresh_beacon_without_live_watcher_stays_alarm() { + local dir out + dir=$(make_guard_case fresh-no-live) + touch "$(case_home "$dir")/state/.last-watcher-beat" + out=$(run_guard_case "$dir") + [ "$(count_text "$out" "WATCHER DOWN - SUPERVISION IS OFF")" -eq 1 ] \ + || fail "a fresh leftover beacon without a live watcher must still alarm: $out" + pass "fm-guard stale banner: a fresh beacon without a live watcher remains unhealthy" +} + +test_x_mode_without_live_watcher_stays_alarm() { + local dir home out + dir=$(make_guard_case x-mode-no-live) + home=$(case_home "$dir") + rm -f "$home/state/task.meta" + : > "$home/state/x-watch.check.sh" + out=$(run_guard_case "$dir") + assert_contains "$out" "X-mode relay polling needs supervision" "X-mode-only need must remain guarded" + pass "fm-guard stale banner: X-mode polling without a live watcher remains unhealthy" +} + test_healthy_recovery_rearms_next_stale_episode() { - local dir home out1 healthy out2 + local dir home out1 healthy out2 pid dir=$(make_guard_case healthy-recovery) home=$(case_home "$dir") out1=$(run_guard_case "$dir") [ "$(count_text "$out1" "WATCHER DOWN - SUPERVISION IS OFF")" -eq 1 ] \ || fail "first stale episode did not print the full banner: $out1" + sleep 60 & + pid=$! + record_live_watcher "$dir" "$pid" || fail "could not record the live watcher for recovery" touch "$home/state/.last-watcher-beat" healthy=$(run_guard_case "$dir") + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true [ -z "$healthy" ] || fail "guard should be silent after watcher recovery, got: $healthy" assert_absent "$home/state/.guard-watcher-stale-banner" \ "healthy recovery must clear the stale-banner marker" rm -f "$home/state/.last-watcher-beat" + rm -rf "$home/state/.watch.lock" out2=$(run_guard_case "$dir") [ "$(count_text "$out2" "WATCHER DOWN - SUPERVISION IS OFF")" -eq 1 ] \ || fail "second stale episode did not re-print the full banner: $out2" @@ -200,15 +238,20 @@ test_read_only_during_episode_observes_without_mutating_marker() { } test_healthy_read_only_does_not_clear_marker() { - local dir home marker before after healthy + local dir home marker before after healthy pid dir=$(make_guard_case healthy-read-only) home=$(case_home "$dir") marker="$home/state/.guard-watcher-stale-banner" run_guard_case "$dir" >/dev/null before=$(cat "$marker") + sleep 60 & + pid=$! + record_live_watcher "$dir" "$pid" || fail "could not record the live watcher for read-only recovery" touch "$home/state/.last-watcher-beat" healthy=$(run_guard_case_read_only "$dir") + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true [ -z "$healthy" ] || fail "healthy read-only guard should stay silent, got: $healthy" assert_present "$marker" "healthy read-only guard must not clear the stale-banner marker" after=$(cat "$marker") @@ -242,6 +285,8 @@ test_read_only_never_mutates_stale_banner_state_files() { test_first_stale_call_prints_full_banner test_repeated_same_episode_prints_reminder_only +test_fresh_beacon_without_live_watcher_stays_alarm +test_x_mode_without_live_watcher_stays_alarm test_healthy_recovery_rearms_next_stale_episode test_concurrent_same_episode_prints_one_full_banner test_home_isolation diff --git a/tests/fm-secondmate-harness.test.sh b/tests/fm-secondmate-harness.test.sh index f8cd7f8077f..e78d6ae5890 100755 --- a/tests/fm-secondmate-harness.test.sh +++ b/tests/fm-secondmate-harness.test.sh @@ -904,6 +904,18 @@ new_world() { printf '%s\n' "$w" } +record_live_watcher_fixture() { + local home=$1 identity + identity=$(FM_STATE_OVERRIDE="$home/state" bash -c '. "$1"; fm_pid_identity "$2"' _ \ + "$ROOT/bin/fm-wake-lib.sh" "$$") || fail "could not identify the live watcher fixture" + mkdir "$home/state/.watch.lock" + printf '%s\n' "$$" > "$home/state/.watch.lock/pid" + printf '%s\n' "$home" > "$home/state/.watch.lock/fm-home" + printf '%s\n' "$ROOT/bin/fm-watch.sh" > "$home/state/.watch.lock/watcher-path" + printf '%s\n' "$identity" > "$home/state/.watch.lock/pid-identity" + touch "$home/state/.last-watcher-beat" +} + # A live secondmate home as a DETACHED worktree of the primary at , with # its seed marker and a live kind=secondmate meta. add_sm_worktree() { @@ -1308,6 +1320,7 @@ test_config_push_propagates_reports_without_ff_or_nudge() { printf 'codex\n' > "$w/home/config/crew-harness" printf 'manual\n' > "$w/home/config/backlog-backend" printf 'tmux\n' > "$w/home/config/backend" + record_live_watcher_fixture "$w/home" err="$w/config-push-basic.err" log="$w/config-push-basic.tmux.log" out=$(run_config_push "$w" "$log" 2>"$err"); status=$? @@ -1492,6 +1505,7 @@ test_config_reread_per_home_changed_sets_and_exact_bytes() { printf '%s\n' "shared secret preference body that must never appear in a config reread" } > "$w/home/data/captain-shared.md" + record_live_watcher_fixture "$w/home" log="$w/config-reread-per-home.tmux.log" err="$w/config-reread-per-home.err" out=$(run_config_push "$w" "$log" 2>"$err"); status=$? diff --git a/tests/fm-supervision-instructions.test.sh b/tests/fm-supervision-instructions.test.sh index e8e5f4f919b..377e95d152a 100755 --- a/tests/fm-supervision-instructions.test.sh +++ b/tests/fm-supervision-instructions.test.sh @@ -51,7 +51,11 @@ test_repair_lines() { out=$(FM_HOME="$home" "$RENDER" --harness claude --queue-pending 1 --repair-line) assert_contains "$out" "After draining queued wakes" "queue-pending prefix missing" - assert_contains "$out" "Claude Code background task" "claude repair line missing background-task mechanism" + assert_contains "$out" "watcher supervision needs Stop-owned automatic recovery" "claude pre-verification repair line is not neutral" + assert_not_contains "$out" "is broken" "claude pre-verification repair line claimed a verified mechanism failure" + assert_not_contains "$out" "FAILED" "claude pre-verification repair line emitted a verified failure notice" + assert_not_contains "$out" "manual background" "claude pre-verification repair line directed a manual background arm" + assert_not_contains "$out" "bin/fm-watch-arm.sh" "claude pre-verification repair line directed an arm command" : > "$home/config/x-mode.env" out=$(FM_HOME="$home" FM_CODEX_WATCH_CHECKPOINT=7 "$RENDER" --harness codex --x-mode 1 --repair-line) @@ -91,8 +95,9 @@ test_cross_harness_ordinary_continuation_and_repair_matrix() { assert_contains "$ordinary" "do not arm another cycle" "claude ordinary-wake line does not forbid a model re-arm" assert_not_contains "$ordinary" "bin/fm-watch-arm.sh" "claude ordinary-wake line incorrectly calls the manual arm" out=$("$RENDER" --harness claude --repair-line) - assert_contains "$out" "Claude Code background task" "claude recovery line lost its tracked background repair" - assert_contains "$out" "bin/fm-watch-arm.sh" "claude recovery line lost the arm command" + assert_contains "$out" "watcher supervision needs Stop-owned automatic recovery" "claude recovery line lost its neutral automatic-recovery guidance" + assert_not_contains "$out" "is broken" "claude recovery line claimed failure before verification" + assert_not_contains "$out" "bin/fm-watch-arm.sh" "claude recovery line must not create a repeatable manual arm loop" out=$("$RENDER" --harness grok) ordinary=$(printf '%s\n' "$out" | grep -F -- '- Ordinary wake:') diff --git a/tests/fm-turnend-guard.test.sh b/tests/fm-turnend-guard.test.sh index 3050d4cd3c2..2518158525f 100755 --- a/tests/fm-turnend-guard.test.sh +++ b/tests/fm-turnend-guard.test.sh @@ -19,7 +19,7 @@ set -u TMP_ROOT=$(fm_test_tmproot fm-turnend-guard) fm_git_identity fmtest fmtest@example.invalid -REQUIRED_REASON='repair missing watcher supervision with bin/fm-watch-arm.sh as its own Claude Code background task' +REQUIRED_REASON='watcher supervision needs Stop-owned automatic recovery; inspect the hook registration and startup status before ending the turn' # --- PREDICATE: bin/fm-supervision-lib.sh ----------------------------------- @@ -283,6 +283,53 @@ test_hook_silent_with_live_lock_and_fresh_beacon() { pass "fm-turnend-guard: silent no-op with a live watcher lock and fresh beacon" } +test_hook_non_claude_health_ignores_claude_budget_contention() { + local dir home pid identity holder harness payload out status + dir=$(make_primary_dir "$TMP_ROOT/hook-non-claude-budget-contention") + home=$(cd "$dir" && pwd) + : > "$dir/state/task1.meta" + sleep 60 & + pid=$! + identity=$(watcher_identity "$dir" "$pid") || { + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + fail "could not identify non-Claude contention watcher" + } + record_watcher_lock "$dir" "$pid" "$identity" + touch "$dir/state/.last-watcher-beat" + printf 'session=claude-episode\ncount=3\nepoch=9\n' > "$dir/state/.turnend-claude-blocks" + printf 'notice-state\n' > "$dir/state/.claude-autoarm-failure-notified" + printf 'alarm-state\n' > "$dir/state/.claude-autoarm-failure-alarmed" + sleep 60 & + holder=$! + mkdir -p "$dir/state/.turnend-claude-blocks.lock" + printf '%s\n' "$holder" > "$dir/state/.turnend-claude-blocks.lock/pid" + while IFS='|' read -r harness payload; do + out=$(printf '%s' "$payload" | FM_HOME="$home" bash "$dir/bin/fm-turnend-guard.sh" 2>&1); status=$? + expect_code 0 "$status" "$harness healthy path must ignore Claude budget-lock contention" + [ -z "$out" ] || fail "$harness healthy path produced output: $out" + [ "$(cat "$dir/state/.turnend-claude-blocks")" = $'session=claude-episode\ncount=3\nepoch=9' ] \ + || fail "$harness healthy path mutated the Claude block budget" + [ "$(cat "$dir/state/.claude-autoarm-failure-notified")" = notice-state ] \ + || fail "$harness healthy path mutated the Claude failure notice" + [ "$(cat "$dir/state/.claude-autoarm-failure-alarmed")" = alarm-state ] \ + || fail "$harness healthy path mutated the Claude attended alarm" + [ "$(cat "$dir/state/.turnend-claude-blocks.lock/pid")" = "$holder" ] \ + || fail "$harness healthy path replaced the Claude budget-lock owner" + done </dev/null || true + wait "$holder" "$pid" 2>/dev/null || true + pass "fm-turnend-guard: healthy non-Claude harness paths ignore Claude episode contention" +} + test_hook_blocks_with_live_lock_and_stale_beacon() { local dir pid identity out status dir=$(make_primary_dir "$TMP_ROOT/hook-live-lock-stale") @@ -340,6 +387,16 @@ test_hook_x_mode_reason_sources_cadence() { pass "fm-turnend-guard: X-mode repair reason sources the cadence config" } +test_hook_x_mode_only_blocks_in_default_mode() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/hook-x-mode-only") + : > "$dir/state/x-watch.check.sh" + out=$(run_hook "$dir" false); status=$? + expect_code 2 "$status" "default hook mode must block an X-mode-only blind turn" + assert_contains "$out" "X-mode relay polling needs supervision" "X-mode-only blind stop must identify its supervision need" + pass "fm-turnend-guard: X-mode-only supervision remains guarded in default mode" +} + test_hook_ignores_repo_state_when_fm_home_set() { local dir home out status dir=$(make_primary_dir "$TMP_ROOT/hook-fm-home-ignore-root") @@ -975,6 +1032,58 @@ run_hook_claude() { printf '{"stop_hook_active":%s,"session_id":"sess-claude-mode"}' "$stop_active" | CLAUDECODE=1 FM_HOME="$home" bash "$dir/bin/fm-turnend-guard.sh" --claude 2>&1 } +seed_claude_failure() { + local dir=$1 outcome=${2:-failed-suppressed} + : > "$dir/state/.claude-autoarm-failure-notified" + printf 'epoch=3 owner_pid=999 outcome=%s updated_at=1\n' "$outcome" > "$dir/state/.claude-autoarm-epoch" + touch -t 202001010000 "$dir/state/.claude-autoarm-epoch" +} + +seed_claude_budget() { + local dir=$1 count=$2 epoch=${3:-2} + printf 'session=sess-claude-mode\ncount=%s\nepoch=%s\n' "$count" "$epoch" > "$dir/state/.turnend-claude-blocks" +} + +record_autoarm_owner() { + local dir=$1 pid=$2 + mkdir -p "$dir/state/.claude-autoarm.lock" + printf '%s\n' "$pid" > "$dir/state/.claude-autoarm.lock/pid" + printf 'autoarm\n' > "$dir/state/.claude-autoarm.lock/role" +} + +install_integrated_autoarm() { + local dir=$1 + cp "$ROOT/bin/fm-claude-stop-autoarm.sh" "$dir/bin/fm-claude-stop-autoarm.sh" + cp "$ROOT/bin/fm-primary-scope-lib.sh" "$dir/bin/fm-primary-scope-lib.sh" + cp "$ROOT/bin/fm-supervision-lib.sh" "$dir/bin/fm-supervision-lib.sh" + cp "$ROOT/bin/fm-wake-lib.sh" "$dir/bin/fm-wake-lib.sh" + cp "$ROOT/bin/fm-session-lock-lib.sh" "$dir/bin/fm-session-lock-lib.sh" + cp "$ROOT/bin/fm-lock.sh" "$dir/bin/fm-lock.sh" + chmod +x "$dir/bin/fm-claude-stop-autoarm.sh" "$dir/bin/fm-lock.sh" + ln -s /bin/bash "$dir/fake-claude" +} + +run_integrated_autoarm() { + local dir=$1 home + home=$(cd "$dir" && pwd) + # shellcheck disable=SC2016 # the fake harness expands FM_HOME inside its child shell. + printf '{"session_id":"sess-claude-mode","stop_hook_active":false}\n' \ + | FM_HOME="$home" "$dir/fake-claude" -c ' + printf "%s\n" "$$" > "$FM_HOME/state/.lock" + "$FM_HOME/bin/fm-claude-stop-autoarm.sh" + ' 2>&1 +} + +write_integrated_failed_arm() { + local dir=$1 + cat > "$dir/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +printf 'watcher: FAILED - persistent fixture failure\n' +exit 1 +SH + chmod +x "$dir/bin/fm-watch-arm.sh" +} + # The 2026-07-21 incident regression: after a spent forced continuation the old # one-shot loop guard ALLOWED a blind stop (stop_hook_active=true) while the # watcher was already dead. In --claude mode the guard must re-block instead. @@ -1001,19 +1110,116 @@ test_hook_claude_mode_reblocks_x_mode_without_tasks() { } test_hook_claude_mode_allows_when_autoarm_owner_alive() { - local dir pid out status + local dir pid out out2 status status2 count count2 dir=$(make_primary_dir "$TMP_ROOT/hook-claude-owner") : > "$dir/state/task1.meta" + seed_claude_failure "$dir" + seed_claude_budget "$dir" 3 sleep 60 & pid=$! - mkdir -p "$dir/state/.claude-autoarm.lock" - printf '%s\n' "$pid" > "$dir/state/.claude-autoarm.lock/pid" + record_autoarm_owner "$dir" "$pid" out=$(run_hook_claude "$dir" false); status=$? + count=$(sed -n '2s/^count=//p' "$dir/state/.turnend-claude-blocks") + out2=$(run_hook_claude "$dir" false); status2=$? + count2=$(sed -n '2s/^count=//p' "$dir/state/.turnend-claude-blocks") kill "$pid" 2>/dev/null || true wait "$pid" 2>/dev/null || true expect_code 0 "$status" "--claude mode must allow when the auto-arm owner process is alive" + expect_code 0 "$status2" "--claude mode must keep allowing the same live auto-arm epoch" [ -z "$out" ] || fail "--claude owner-claimed allow produced output: $out" - pass "fm-turnend-guard --claude: allows the stop when the Stop auto-arm owner holds this home" + [ -z "$out2" ] || fail "repeated same-owner allow produced output: $out2" + [ "$count" = 4 ] || fail "new live auto-arm epoch did not advance failure progression from 3 to 4: $count" + [ "$count2" = 4 ] || fail "repeated observation advanced the same auto-arm epoch twice: $count2" + assert_present "$dir/state/.claude-autoarm-failure-notified" "live auto-arm owner cleared the failure episode" + assert_absent "$dir/state/.claude-autoarm-failure-alarmed" "live automatic continuation emitted the attended fail-open alarm" + pass "fm-turnend-guard --claude: a live arming epoch advances once and repeated observation is idempotent" +} + +test_hook_claude_mode_repeated_failed_to_arming_interleavings_reach_fail_open() { + local dir out status pid i count epoch + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-arming-interleavings") + : > "$dir/state/task1.meta" + : > "$dir/state/.claude-autoarm-failure-notified" + printf 'epoch=3 owner_pid=999 outcome=failed updated_at=%s\n' "$(date +%s)" > "$dir/state/.claude-autoarm-epoch" + out=$(run_hook_claude "$dir" true); status=$? + expect_code 0 "$status" "the first verified failed epoch must own its automatic handoff" + + epoch=3 + for i in 1 2 3 4; do + epoch=$((epoch + 1)) + sleep 60 & + pid=$! + record_autoarm_owner "$dir" "$pid" + printf 'epoch=%s owner_pid=%s outcome=arming updated_at=%s\n' "$epoch" "$pid" "$(date +%s)" > "$dir/state/.claude-autoarm-epoch" + out=$(run_hook_claude "$dir" true); status=$? + expect_code 0 "$status" "active arming epoch $i must own its Stop while advancing the failure budget" + count=$(sed -n '2s/^count=//p' "$dir/state/.turnend-claude-blocks") + [ "$count" = "$i" ] || fail "arming epoch $i produced non-monotonic count $count" + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + rm -rf "$dir/state/.claude-autoarm.lock" + epoch=$((epoch + 1)) + printf 'epoch=%s owner_pid=999 outcome=failed-suppressed updated_at=%s\n' "$epoch" "$(date +%s)" > "$dir/state/.claude-autoarm-epoch" + done + + out=$(run_hook_claude "$dir" true); status=$? + expect_code 0 "$status" "repeated failed-to-arming interleavings must reach terminal fail-open" + assert_contains "$out" 'FIRSTMATE SUPERVISION IS GENUINELY DOWN' "arming interleavings stalled before the bounded fail-open" + assert_present "$dir/state/.claude-autoarm-failure-alarmed" "arming interleavings did not consume the one-time alarm" + pass "fm-turnend-guard --claude: repeated failed-to-arming races make bounded monotonic progress" +} + +test_hook_claude_mode_terminal_boundary_excludes_starting_owner() { + local dir fakebin ready release once guard_out guard_status auto_out auto_status guard_pid + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-terminal-boundary") + : > "$dir/state/task1.meta" + : > "$dir/state/.claude-autoarm-failure-notified" + printf 'epoch=3 owner_pid=999 outcome=failed-suppressed updated_at=%s\n' "$(date +%s)" > "$dir/state/.claude-autoarm-epoch" + seed_claude_budget "$dir" 4 3 + install_integrated_autoarm "$dir" + write_integrated_failed_arm "$dir" + fakebin="$dir/fakebin" + ready="$dir/terminal-ready" + release="$dir/terminal-release" + once="$dir/terminal-once" + guard_out="$dir/guard.out" + guard_status="$dir/guard.status" + mkdir -p "$fakebin" + mkfifo "$ready" "$release" + cat > "$fakebin/cat" <<'SH' +#!/usr/bin/env bash +if [ "$1" = "$FM_TERMINAL_ROLE_PATH" ] \ + && [ "$(/bin/cat "$1" 2>/dev/null || true)" = terminal-check ] \ + && (set -C; : > "$FM_TERMINAL_ONCE") 2>/dev/null; then + printf 'ready\n' > "$FM_TERMINAL_READY" + IFS= read -r _ < "$FM_TERMINAL_RELEASE" +fi +exec /bin/cat "$@" +SH + chmod +x "$fakebin/cat" + ( + printf '{"stop_hook_active":true,"session_id":"sess-claude-mode"}' \ + | PATH="$fakebin:$PATH" \ + FM_TERMINAL_ROLE_PATH="$dir/state/.claude-autoarm.lock/role" \ + FM_TERMINAL_READY="$ready" \ + FM_TERMINAL_RELEASE="$release" \ + FM_TERMINAL_ONCE="$once" \ + CLAUDECODE=1 FM_HOME="$dir" bash "$dir/bin/fm-turnend-guard.sh" --claude \ + > "$guard_out" 2>&1 + printf '%s\n' "$?" > "$guard_status" + ) & + guard_pid=$! + IFS= read -r _ < "$ready" + auto_out=$(run_integrated_autoarm "$dir"); auto_status=$? + printf 'release\n' > "$release" + wait "$guard_pid" + expect_code 0 "$auto_status" "an owner starting inside the terminal window must lose the existing owner boundary" + [ -z "$auto_out" ] || fail "excluded terminal-window owner produced output: $auto_out" + assert_absent "$dir/state/arm-ran" "excluded terminal-window owner started an arm cycle" + expect_code 0 "$(cat "$guard_status")" "terminal boundary guard must complete without deadlock" + assert_contains "$(cat "$guard_out")" 'FIRSTMATE SUPERVISION IS GENUINELY DOWN' "terminal boundary did not produce the one-time alarm" + assert_absent "$dir/state/.claude-autoarm.lock" "terminal boundary left its owner lock behind" + pass "fm-turnend-guard --claude: terminal owner boundary excludes a concurrent start without deadlock" } test_hook_claude_mode_allows_on_fresh_rewake_epoch() { @@ -1027,6 +1233,162 @@ test_hook_claude_mode_allows_on_fresh_rewake_epoch() { pass "fm-turnend-guard --claude: fresh rewake epoch prevents a duplicate continuation for the same event" } +test_hook_claude_mode_preserves_fresh_failed_progression() { + local dir out status count + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-failed-epoch") + : > "$dir/state/task1.meta" + : > "$dir/state/.claude-autoarm-failure-notified" + printf 'epoch=3 owner_pid=999 outcome=failed updated_at=%s\n' "$(date +%s)" > "$dir/state/.claude-autoarm-epoch" + out=$(run_hook_claude "$dir" true); status=$? + expect_code 0 "$status" "the first fresh failed epoch must count as its automatic continuation" + [ -z "$out" ] || fail "fresh failed-epoch allow produced output: $out" + assert_present "$dir/state/.turnend-claude-blocks" "fresh failed epoch did not preserve bounded progression" + count=$(sed -n '2s/^count=//p' "$dir/state/.turnend-claude-blocks") + [ "$count" = 0 ] || fail "the owned first failed epoch must not consume a blocked-stop count, got $count" + printf 'epoch=4 owner_pid=999 outcome=failed-suppressed updated_at=%s\n' "$(date +%s)" > "$dir/state/.claude-autoarm-epoch" + out=$(run_hook_claude "$dir" true); status=$? + expect_code 2 "$status" "a later fresh failed epoch must consume the bounded progression" + assert_absent "$dir/state/.claude-autoarm-failure-alarmed" "fresh failure progression emitted the attended fail-open alarm too early" + count=$(sed -n '2s/^count=//p' "$dir/state/.turnend-claude-blocks") + [ "$count" = 1 ] || fail "the later failed epoch must advance the blocked-stop count, got $count" + pass "fm-turnend-guard --claude: fresh failed epochs preserve and advance monotonic fail-open progression" +} + +test_hook_claude_mode_integrated_monotonic_fail_open() { + local dir out status guard_out guard_status i pid identity count + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-integrated-fail-open") + : > "$dir/state/task1.meta" + install_integrated_autoarm "$dir" + write_integrated_failed_arm "$dir" + + out=$(run_integrated_autoarm "$dir"); status=$? + expect_code 2 "$status" "the first exhausted auto-arm cycle must emit its one failure notice" + assert_contains "$out" "automatic supervision mechanism is broken" "the first integrated failure notice is missing" + guard_out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" true); guard_status=$? + expect_code 0 "$guard_status" "the first failed epoch must own its Stop handoff" + count=$(sed -n '2s/^count=//p' "$dir/state/.turnend-claude-blocks") + [ "$count" = 0 ] || fail "the first owned failure epoch must preserve a zero blocked-stop count, got $count" + + for i in 1 2 3 4; do + out=$(run_integrated_autoarm "$dir"); status=$? + expect_code 2 "$status" "failed epoch $i must retain the automatic retry handoff" + [ -z "$out" ] || fail "failed epoch $i repeated the operator notice: $out" + guard_out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" true); guard_status=$? + if [ "$i" -lt 4 ]; then + expect_code 2 "$guard_status" "failed epoch $i must consume a bounded blind-stop block" + assert_not_contains "$guard_out" 'FIRSTMATE SUPERVISION IS GENUINELY DOWN' "fail-open fired before the bounded progression ended" + else + expect_code 0 "$guard_status" "the bounded failure progression must reach the attended fail-open" + assert_contains "$guard_out" 'FIRSTMATE SUPERVISION IS GENUINELY DOWN' "the integrated fail-open alarm is missing" + assert_present "$dir/state/.claude-autoarm-failure-alarmed" "the integrated fail-open did not consume its episode alarm" + fi + done + + out=$(run_integrated_autoarm "$dir"); status=$? + expect_code 0 "$status" "the auto-arm must not re-trigger continuation after the final fail-open" + [ -z "$out" ] || fail "post-fail-open auto-arm produced continuation output: $out" + guard_out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" true); guard_status=$? + expect_code 2 "$guard_status" "a later unhealthy stop in the same episode must remain attended" + assert_not_contains "$guard_out" 'FIRSTMATE SUPERVISION IS GENUINELY DOWN' "the attended alarm repeated in the same episode" + + sleep 60 & + pid=$! + identity=$(watcher_identity "$dir" "$pid") || { + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + fail "could not identify the positive recovery watcher" + } + record_watcher_lock "$dir" "$pid" "$identity" + touch "$dir/state/.last-watcher-beat" + out=$(run_integrated_autoarm "$dir"); status=$? + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + rm -rf "$dir/state/.watch.lock" + expect_code 0 "$status" "positive watcher recovery must make the auto-arm silent" + assert_absent "$dir/state/.claude-autoarm-failure-notified" "positive recovery left the failure notice marker" + assert_absent "$dir/state/.claude-autoarm-failure-alarmed" "positive recovery left the attended alarm marker" + assert_absent "$dir/state/.turnend-claude-blocks" "positive recovery left the bounded block budget" + guard_out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" false); guard_status=$? + expect_code 2 "$guard_status" "a guard after one-shot recovery must start a fresh failure budget" + count=$(sed -n '2s/^count=//p' "$dir/state/.turnend-claude-blocks") + [ "$count" = 1 ] || fail "the independent post-recovery failure must start at count 1, got $count" + + out=$(run_integrated_autoarm "$dir"); status=$? + expect_code 2 "$status" "a later failure after positive recovery must start a new episode" + assert_contains "$out" "automatic supervision mechanism is broken" "the new failure episode notice was suppressed" + pass "fm-turnend-guard --claude: integrated fresh failures reach one bounded fail-open, stop continuation, and reset on recovery" +} + +test_hook_claude_mode_recovery_contention_is_not_ordinary_allow() { + local dir pid identity holder out status + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-recovery-contention") + : > "$dir/state/task1.meta" + seed_claude_budget "$dir" 3 + : > "$dir/state/.claude-autoarm-failure-notified" + : > "$dir/state/.claude-autoarm-failure-alarmed" + sleep 60 & + pid=$! + identity=$(watcher_identity "$dir" "$pid") || fail "could not identify recovery-contention watcher" + record_watcher_lock "$dir" "$pid" "$identity" + touch "$dir/state/.last-watcher-beat" + sleep 60 & + holder=$! + mkdir -p "$dir/state/.turnend-claude-blocks.lock" + printf '%s\n' "$holder" > "$dir/state/.turnend-claude-blocks.lock/pid" + out=$(run_hook_claude "$dir" false); status=$? + expect_code 2 "$status" "a healthy guard must continue when the episode reset lock is busy" + [ -z "$out" ] || fail "guard recovery contention produced output: $out" + assert_present "$dir/state/.turnend-claude-blocks" "guard contention partially cleared the block budget" + assert_present "$dir/state/.claude-autoarm-failure-notified" "guard contention partially cleared the failure notice" + assert_present "$dir/state/.claude-autoarm-failure-alarmed" "guard contention partially cleared the attended alarm" + kill "$holder" 2>/dev/null || true + wait "$holder" 2>/dev/null || true + out=$(run_hook_claude "$dir" false); status=$? + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + expect_code 0 "$status" "the healthy guard must allow after completing the episode reset" + assert_absent "$dir/state/.turnend-claude-blocks" "successful guard reset left the block budget" + assert_absent "$dir/state/.claude-autoarm-failure-notified" "successful guard reset left the failure notice" + assert_absent "$dir/state/.claude-autoarm-failure-alarmed" "successful guard reset left the attended alarm" + pass "fm-turnend-guard --claude: reset contention preserves all episode state until retry" +} + +test_hook_claude_mode_concurrent_recovery_resets_are_idempotent() { + local dir pid identity auto_pid guard_pid auto_status guard_status + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-concurrent-recovery") + : > "$dir/state/task1.meta" + install_integrated_autoarm "$dir" + write_integrated_failed_arm "$dir" + seed_claude_budget "$dir" 3 + : > "$dir/state/.claude-autoarm-failure-notified" + : > "$dir/state/.claude-autoarm-failure-alarmed" + sleep 60 & + pid=$! + identity=$(watcher_identity "$dir" "$pid") || fail "could not identify concurrent recovery watcher" + record_watcher_lock "$dir" "$pid" "$identity" + touch "$dir/state/.last-watcher-beat" + (run_integrated_autoarm "$dir" > "$dir/auto.out"; printf '%s\n' "$?" > "$dir/auto.status") & + auto_pid=$! + (FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" false > "$dir/guard.out"; printf '%s\n' "$?" > "$dir/guard.status") & + guard_pid=$! + wait "$auto_pid" + wait "$guard_pid" + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + auto_status=$(cat "$dir/auto.status") + guard_status=$(cat "$dir/guard.status") + case "$auto_status:$guard_status" in + 0:0|0:2|2:0) : ;; + *) fail "concurrent reset callers returned unsafe statuses auto=$auto_status guard=$guard_status" ;; + esac + assert_absent "$dir/state/.turnend-claude-blocks" "concurrent recovery left the block budget" + assert_absent "$dir/state/.claude-autoarm-failure-notified" "concurrent recovery left the failure notice" + assert_absent "$dir/state/.claude-autoarm-failure-alarmed" "concurrent recovery left the attended alarm" + assert_absent "$dir/state/.claude-autoarm.lock" "concurrent recovery left the owner lock" + assert_absent "$dir/state/.turnend-claude-blocks.lock" "concurrent recovery left the budget lock" + pass "fm-turnend-guard --claude: concurrent auto-arm and guard resets are idempotent and deadlock-free" +} + test_hook_claude_mode_stale_rewake_epoch_blocks() { local dir out status dir=$(make_primary_dir "$TMP_ROOT/hook-claude-stale-epoch") @@ -1038,21 +1400,69 @@ test_hook_claude_mode_stale_rewake_epoch_blocks() { pass "fm-turnend-guard --claude: stale rewake epoch does not allow a blind stop" } -test_hook_claude_mode_block_budget_then_degraded_allow() { +test_hook_claude_mode_budget_without_verified_failure_keeps_blocking() { local dir out status i dir=$(make_primary_dir "$TMP_ROOT/hook-claude-budget") : > "$dir/state/task1.meta" - for i in 1 2 3; do + for i in 1 2 3 4; do out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" false); status=$? expect_code 2 "$status" "--claude block $i must exit 2 within the budget" done + assert_not_contains "$out" 'systemMessage' "budget exhaustion without verified auto-arm failure must not fail open" + assert_absent "$dir/state/.claude-autoarm-failure-alarmed" "unverified budget exhaustion recorded an attended alarm" + pass "fm-turnend-guard --claude: budget exhaustion alone cannot permit a blind stop" +} + +test_hook_claude_mode_verified_failure_alarm_is_loud_and_once() { + local dir out out2 status status2 + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-verified-alarm") + : > "$dir/state/task1.meta" + seed_claude_failure "$dir" + seed_claude_budget "$dir" 3 out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" true); status=$? - expect_code 0 "$status" "--claude must allow degraded once the consecutive-block budget is exhausted" - assert_contains "$out" '"systemMessage"' "--claude degraded allow must surface a visible systemMessage" - assert_contains "$out" 'block budget exhausted' "--claude degraded allow must name the exhausted budget" - out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" false); status=$? - expect_code 2 "$status" "--claude budget must reset after the degraded allow so the next chain re-engages" - pass "fm-turnend-guard --claude: re-block budget stays below the 8-block cap and resets after degraded allow" + expect_code 0 "$status" "verified failure with exhausted budget must take the bounded attended fail-open" + assert_contains "$out" 'FIRSTMATE SUPERVISION IS GENUINELY DOWN' "bounded fail-open alarm was not unmistakable" + assert_contains "$out" 'Keep this session attended' "bounded fail-open alarm omitted the attended-session action" + assert_contains "$out" 'diagnose the automatic Stop-hook and watcher startup' "bounded fail-open alarm omitted automatic-mechanism diagnosis" + assert_not_contains "$out" 'fm-watch-arm.sh' "bounded fail-open alarm assigned a manual watcher launch" + assert_present "$dir/state/.claude-autoarm-failure-alarmed" "bounded fail-open did not consume the episode alarm" + out2=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" true); status2=$? + expect_code 2 "$status2" "a consumed attended alarm must make later unhealthy stops block again" + assert_not_contains "$out2" 'FIRSTMATE SUPERVISION IS GENUINELY DOWN' "attended failure alarm repeated in one episode" + pass "fm-turnend-guard --claude: verified fail-open is loud, bounded, attended, and non-repeating" +} + +test_hook_claude_mode_fail_open_requires_notice_and_failure_epoch() { + local no_notice notice_only out status + no_notice=$(make_primary_dir "$TMP_ROOT/hook-claude-alarm-no-notice") + : > "$no_notice/state/task1.meta" + printf 'epoch=3 owner_pid=999 outcome=failed-suppressed updated_at=1\n' > "$no_notice/state/.claude-autoarm-epoch" + touch -t 202001010000 "$no_notice/state/.claude-autoarm-epoch" + seed_claude_budget "$no_notice" 3 + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$no_notice" true); status=$? + expect_code 2 "$status" "an exhausted failure epoch without the consumed notice must remain blocking" + + notice_only=$(make_primary_dir "$TMP_ROOT/hook-claude-alarm-no-epoch") + : > "$notice_only/state/task1.meta" + : > "$notice_only/state/.claude-autoarm-failure-notified" + seed_claude_budget "$notice_only" 3 + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$notice_only" true); status=$? + expect_code 2 "$status" "a consumed notice without an exhausted failure epoch must remain blocking" + pass "fm-turnend-guard --claude: fail-open requires both exhausted retries and consumed notice" +} + +test_hook_claude_mode_away_mode_never_uses_stop_autoarm_fail_open() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/hook-claude-alarm-afk") + : > "$dir/state/task1.meta" + : > "$dir/state/.afk" + seed_claude_failure "$dir" + seed_claude_budget "$dir" 3 + out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" true); status=$? + expect_code 2 "$status" "away mode must not use a stale Stop-autoarm failure to fail open" + assert_contains "$out" 'Away mode owns watcher supervision' "away-mode block lost its daemon ownership guidance" + assert_absent "$dir/state/.claude-autoarm-failure-alarmed" "away mode consumed the Stop-autoarm attended alarm" + pass "fm-turnend-guard --claude: away ownership excludes the Stop-autoarm fail-open" } test_hook_claude_mode_allow_resets_budget() { @@ -1062,6 +1472,8 @@ test_hook_claude_mode_allow_resets_budget() { out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" false); status=$? expect_code 2 "$status" "first --claude block must exit 2" [ -f "$dir/state/.turnend-claude-blocks" ] || fail "--claude block must record the consecutive-block budget" + : > "$dir/state/.claude-autoarm-failure-notified" + : > "$dir/state/.claude-autoarm-failure-alarmed" sleep 60 & pid=$! identity=$(watcher_identity "$dir" "$pid") || { @@ -1077,9 +1489,11 @@ test_hook_claude_mode_allow_resets_budget() { rm -rf "$dir/state/.watch.lock" expect_code 0 "$status" "--claude must allow once the watcher is healthy again" [ ! -f "$dir/state/.turnend-claude-blocks" ] || fail "--claude allow must reset the consecutive-block budget" + [ ! -f "$dir/state/.claude-autoarm-failure-notified" ] || fail "positive watcher recovery must reset the failure notice" + [ ! -f "$dir/state/.claude-autoarm-failure-alarmed" ] || fail "positive watcher recovery must reset the attended alarm" out=$(FM_CLAUDE_AUTOARM_SYNC_WAIT_MS=100 run_hook_claude "$dir" false); status=$? expect_code 2 "$status" "a later unhealthy chain must re-block from a fresh budget" - pass "fm-turnend-guard --claude: any allow resets the consecutive-block budget" + pass "fm-turnend-guard --claude: positive watcher recovery resets failure episode state" } test_hook_claude_mode_waits_for_late_claim() { @@ -1088,9 +1502,8 @@ test_hook_claude_mode_waits_for_late_claim() { : > "$dir/state/task1.meta" ( sleep 0.4 - mkdir -p "$dir/state/.claude-autoarm.lock" sleep 60 & - printf '%s\n' $! > "$dir/state/.claude-autoarm.lock/pid" + record_autoarm_owner "$dir" $! printf '%s\n' $! > "$dir/holder.pid" wait ) & @@ -1114,8 +1527,7 @@ test_hook_claude_mode_secondmate_reblocks_like_primary() { assert_contains "$out" "TURN WOULD END BLIND" "--claude secondmate re-block must carry the blind-turn banner" sleep 60 & pid=$! - mkdir -p "$dir/state/.claude-autoarm.lock" - printf '%s\n' "$pid" > "$dir/state/.claude-autoarm.lock/pid" + record_autoarm_owner "$dir" "$pid" out=$(run_hook_claude "$dir" false); status=$? kill "$pid" 2>/dev/null || true wait "$pid" 2>/dev/null || true @@ -1135,10 +1547,12 @@ test_hook_blocks_when_fresh_beacon_has_no_live_lock test_hook_blocks_source_only_home test_hook_blocks_when_dead_lock_has_fresh_beacon test_hook_silent_with_live_lock_and_fresh_beacon +test_hook_non_claude_health_ignores_claude_budget_contention test_hook_blocks_with_live_lock_and_stale_beacon test_hook_blocks_when_unhealthy_in_primary test_hook_blocks_from_fm_home_state test_hook_x_mode_reason_sources_cadence +test_hook_x_mode_only_blocks_in_default_mode test_hook_ignores_repo_state_when_fm_home_set test_hook_uses_state_override test_hook_loop_guard_allows_retry @@ -1169,9 +1583,18 @@ test_pi_extension_retries_after_followup_delivery_failure test_hook_claude_mode_reblocks_stop_hook_active_when_unhealthy test_hook_claude_mode_reblocks_x_mode_without_tasks test_hook_claude_mode_allows_when_autoarm_owner_alive +test_hook_claude_mode_repeated_failed_to_arming_interleavings_reach_fail_open +test_hook_claude_mode_terminal_boundary_excludes_starting_owner test_hook_claude_mode_allows_on_fresh_rewake_epoch +test_hook_claude_mode_preserves_fresh_failed_progression +test_hook_claude_mode_integrated_monotonic_fail_open +test_hook_claude_mode_recovery_contention_is_not_ordinary_allow +test_hook_claude_mode_concurrent_recovery_resets_are_idempotent test_hook_claude_mode_stale_rewake_epoch_blocks -test_hook_claude_mode_block_budget_then_degraded_allow +test_hook_claude_mode_budget_without_verified_failure_keeps_blocking +test_hook_claude_mode_verified_failure_alarm_is_loud_and_once +test_hook_claude_mode_fail_open_requires_notice_and_failure_epoch +test_hook_claude_mode_away_mode_never_uses_stop_autoarm_fail_open test_hook_claude_mode_allow_resets_budget test_hook_claude_mode_waits_for_late_claim test_hook_claude_mode_secondmate_reblocks_like_primary diff --git a/tests/fm-wake-queue.test.sh b/tests/fm-wake-queue.test.sh index 569f18b42f0..b86eb9ac64e 100755 --- a/tests/fm-wake-queue.test.sh +++ b/tests/fm-wake-queue.test.sh @@ -216,9 +216,9 @@ test_drain_dedupes_obvious_duplicates() { # watcher liveness via fm-guard.sh: a lapsed re-arm chain then surfaces even on a # plain drain-and-handle turn that runs no other supervision script. It must warn # when work is in flight with no live watcher, and stay silent right after a -# normal fire (a fresh beacon within grace), so it never false-alarms every wake. +# normal fire from a live watcher with a fresh beacon, so it never false-alarms. test_drain_asserts_watcher_liveness() { - local dir state err + local dir state err identity dir=$(make_case drain-liveness) state="$dir/state" err="$dir/drain.err" @@ -226,12 +226,20 @@ test_drain_asserts_watcher_liveness() { FM_STATE_OVERRIDE="$state" "$DRAIN" >/dev/null 2> "$err" || fail "drain failed while asserting liveness" grep -F 'WATCHER DOWN' "$err" >/dev/null || fail "drain did not surface the watcher-down banner with work in flight and no live watcher" : > "$err" + identity=$(FM_STATE_OVERRIDE="$state" bash -c '. "$1"; fm_pid_identity "$2"' _ "$ROOT/bin/fm-wake-lib.sh" "$$") \ + || fail "could not identify the live watcher fixture" + mkdir "$state/.watch.lock" + printf '%s\n' "$$" > "$state/.watch.lock/pid" + printf '%s\n' "$dir" > "$state/.watch.lock/fm-home" + printf '%s\n' "$WATCH" > "$state/.watch.lock/watcher-path" + printf '%s\n' "$identity" > "$state/.watch.lock/pid-identity" touch "$state/.last-watcher-beat" - FM_STATE_OVERRIDE="$state" FM_GUARD_GRACE=300 "$DRAIN" >/dev/null 2> "$err" || fail "drain failed with a fresh beacon" + FM_HOME="$dir" FM_STATE_OVERRIDE="$state" FM_GUARD_GRACE=300 "$DRAIN" >/dev/null 2> "$err" \ + || fail "drain failed with a live watcher and fresh beacon" if grep -F 'WATCHER DOWN' "$err" >/dev/null; then - fail "drain false-alarmed right after a normal fire (fresh beacon within grace)" + fail "drain false-alarmed with a live watcher and fresh beacon" fi - pass "drain asserts watcher liveness: warns on a lapse, stays silent right after a fire" + pass "drain asserts watcher liveness: warns on a lapse, stays silent for a live watcher with a fresh beacon" } test_structural_signal_enrichment_preserves_raw_rows() { diff --git a/tests/fm-watcher-lock.test.sh b/tests/fm-watcher-lock.test.sh index e741ec21e8e..4ffd4262bcc 100755 --- a/tests/fm-watcher-lock.test.sh +++ b/tests/fm-watcher-lock.test.sh @@ -115,7 +115,7 @@ test_guard_warnings() { # warning follows it, and the guidance is repair-after-drain (never the # old conflicting "restart NOW first"). # (2) a fresh watcher and an empty queue: total silence. - local dir state err first banner_line queue_line + local dir state err first banner_line queue_line pid identity dir=$(make_case guard) state="$dir/state" err="$dir/guard.err" @@ -138,9 +138,9 @@ test_guard_warnings() { grep -F 'last beat: never' "$err" >/dev/null || fail "guard banner missing the beacon age" grep -F 'guarded operation WILL still run' "$err" >/dev/null || fail "guard banner missing generic continuation wording" ! grep -F 'requested message WILL still be sent' "$err" >/dev/null || fail "shared guard used send-specific continuation wording" - grep -F 'repair missing watcher supervision' "$err" >/dev/null || fail "guard banner missing the harness-aware fix command" + grep -F 'watcher supervision needs Stop-owned automatic recovery' "$err" >/dev/null || fail "guard banner missing neutral automatic-recovery guidance" grep -F 'queued wakes pending - drain them' "$err" >/dev/null || fail "guard did not warn about pending queue" - grep -F 'After draining queued wakes, repair missing watcher supervision' "$err" >/dev/null || fail "guard did not order supervision repair after drain" + grep -F 'After draining queued wakes, watcher supervision needs Stop-owned automatic recovery' "$err" >/dev/null || fail "guard did not order neutral automatic recovery after drain" ! grep -F 'Restart it NOW, before anything else' "$err" >/dev/null || fail "guard still gave conflicting restart-first instruction" ! grep -F 'as the harness-tracked background task' "$err" >/dev/null || fail "guard still printed the old universal background-task repair text" banner_line=$(grep -n 'WATCHER DOWN' "$err" | head -1 | cut -d: -f1) @@ -156,17 +156,27 @@ test_guard_warnings() { CLAUDECODE=1 PI_CODING_AGENT='' GROK_AGENT='' FM_ROOT_OVERRIDE="$dir" FM_STATE_OVERRIDE="$state" FM_GUARD_GRACE=1 "$ROOT/bin/fm-guard.sh" 2> "$err" >/dev/null || fail "guard failed" grep -F "source '$dir/config/x-mode.env' first" "$err" >/dev/null || fail "guard repair line did not source the X-mode cadence config" - # (2) fresh watcher, empty queue -> silence. + # (2) live watcher plus fresh beacon, empty queue -> silence. dir=$(make_case guard-fresh) state="$dir/state" err="$dir/guard.err" printf 'project=x\n' > "$state/task.meta" + sleep 60 & + pid=$! + identity=$(FM_STATE_OVERRIDE="$state" bash -c '. "$1"; fm_pid_identity "$2"' _ "$LIB" "$pid") || fail "could not identify fresh guard watcher" + mkdir -p "$state/.watch.lock" + printf '%s\n' "$pid" > "$state/.watch.lock/pid" + printf '%s\n' "$dir" > "$state/.watch.lock/fm-home" + printf '%s\n' "$WATCH" > "$state/.watch.lock/watcher-path" + printf '%s\n' "$identity" > "$state/.watch.lock/pid-identity" touch "$state/.last-watcher-beat" # Non-git FM_ROOT keeps the worktree-tangle check inert so "fresh watcher -> # total silence" stays a pure assertion about watcher state. FM_ROOT_OVERRIDE="$dir" FM_STATE_OVERRIDE="$state" FM_GUARD_GRACE=300 "$ROOT/bin/fm-guard.sh" 2> "$err" >/dev/null || fail "guard failed" - [ ! -s "$err" ] || fail "guard warned with a fresh watcher and no queued wakes: $(cat "$err")" - pass "guard banner leads when down with pending wakes (repair-after-drain) and stays silent when fresh" + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + [ ! -s "$err" ] || fail "guard warned with a live watcher and fresh beacon: $(cat "$err")" + pass "guard banner leads when down with pending wakes (repair-after-drain) and stays silent when live and fresh" } test_lock_single_winner_under_concurrency() { diff --git a/tests/fm-x-mode.test.sh b/tests/fm-x-mode.test.sh index baed0b28d4a..24cdb60393c 100755 --- a/tests/fm-x-mode.test.sh +++ b/tests/fm-x-mode.test.sh @@ -888,7 +888,8 @@ test_bootstrap_opt_out_cleanup() { printf 'FMX_PAIRING_TOKEN=\n' > "$home/.env" out=$(CLAUDECODE=1 FM_HOME="$home" "$ROOT/bin/fm-bootstrap.sh" 2>/dev/null) assert_contains "$out" "FMX: X mode off" "opt-out must announce X mode off when it removed artifacts" - assert_contains "$out" "Claude Code background task" "opt-out remediation must use the harness-aware repair renderer" + assert_contains "$out" "watcher supervision needs Stop-owned automatic recovery" "opt-out remediation must use neutral automatic-recovery guidance" + assert_not_contains "$out" "is broken" "opt-out remediation claimed an unverified mechanism failure" assert_not_contains "$out" "bin/fm-watch-arm.sh --restart" "opt-out remediation must not hardcode a background-arm restart" assert_absent "$home/state/x-watch.check.sh" "opt-out must remove the shim" assert_absent "$home/config/x-mode.env" "opt-out must remove the cadence config" From 4ee4a0a2790cfaa5e47b30fa462f16546f2ab5b6 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Sun, 2 Aug 2026 22:06:39 -0700 Subject: [PATCH 03/35] feat(bin): require an explicit per-task delivery contract (#1563) * feat(bin): require an explicit ship delivery mode in fm-brief A ship brief's definition of done was shaped by a silent per-project registry lookup, so an adjusted brief and the task's recorded delivery could disagree and no one had to decide anything per task. fm-brief now requires --mode on ship scaffolds, validates it against the closed set, refuses the conditional no-mistakes-prod-only registry policy as a task mode, and records the choice as a fixed machine-readable "Delivery contract: mode=" line that fm-spawn can check. --mode is refused on scout and secondmate scaffolds, and --yolo is refused outright because the worker never owns approval decisions. * feat(bin): require an explicit ship delivery contract at spawn and promotion fm-spawn resolved every ship and scout task's mode and yolo from the project registry, so the delivery posture was never a per-task decision and could contradict the brief the worker was about to follow. fm-spawn now requires --mode and --yolo on ship spawns, validates both against their closed sets, and reads the brief's recorded delivery contract line and refuses a mismatch before any endpoint exists; a brief scaffolded before that line existed warns once and launches on the flag. A batch carries one shared contract that each pair still checks against its own brief. Scout and secondmate spawns refuse the flags, and a scout now records no mode or yolo at all, which teardown and the snapshot already tolerate. When the explicit mode carries less rigor than the project's standing posture, a deviation notice is printed and the spawn continues, so the registry stays advisory rather than an enforced default. fm-promote requires the same two flags, because a scout carries no posture to inherit, and writes them into the task record with the kind flip. fm-project-mode keeps its one registry parser for the mechanical consumers that have no task in hand, accepts the conditional no-mistakes-prod-only annotation and maps it to its most rigorous leg for them, and grows --raw so the deviation notice can tell a conditional policy apart from a flat mode. * docs: record the explicit per-task delivery contract AGENTS.md section 7 now owns how each ship task's mode and yolo are resolved at intake, including the surface classification for a no-mistakes-prod-only project and the unregistered-project fallback, and the project-management skill defines that conditional policy as a registration-time posture with its defaults and initialization consequences. The registry blurb, script table, and architecture section follow: the registry records the captain's standing posture, and task delivery is decided per task and passed explicitly. * test: pass ship delivery flags per call site in the Herdr launcher e2e The shared spawn helper also launches a secondmate, which refuses the flags, so the contract belongs at each ship call site rather than inside the helper. * test: pass the ship delivery contract in the secondmate suites Both suites scaffold or spawn an ordinary ship task as the control case for a secondmate assertion, so each needs the explicit contract the ship path now requires. --- .agents/skills/project-management/SKILL.md | 25 +- AGENTS.md | 8 +- bin/fm-brief.sh | 86 ++++-- bin/fm-project-mode.sh | 43 ++- bin/fm-promote.sh | 73 ++++- bin/fm-spawn.sh | 143 +++++++-- bin/fm-test-run.sh | 3 +- docs/architecture.md | 9 +- docs/scripts.md | 6 +- tests/fm-ask-user-authority.test.sh | 2 +- tests/fm-backend-autodetect-smoke.test.sh | 2 +- ...ckend-herdr-launcher-workspace-e2e.test.sh | 16 +- .../fm-backend-herdr-presentation-e2e.test.sh | 2 +- ...ckend-herdr-workspace-per-home-e2e.test.sh | 4 +- tests/fm-backend-orca.test.sh | 14 +- tests/fm-backend.test.sh | 14 +- tests/fm-brief.test.sh | 111 ++++++- tests/fm-busy-adapter-wiring.test.sh | 4 + tests/fm-gate-refuse.test.sh | 2 +- tests/fm-grok-harness.test.sh | 2 +- tests/fm-kimi-harness.test.sh | 2 +- tests/fm-secondmate-harness.test.sh | 2 +- tests/fm-secondmate-safety.test.sh | 2 +- tests/fm-spawn-batch.test.sh | 49 ++- tests/fm-spawn-dispatch-profile.test.sh | 58 ++-- tests/fm-spawn-worktree-settle.test.sh | 2 +- tests/fm-tangle-guard.test.sh | 6 +- tests/fm-task-delivery.test.sh | 282 ++++++++++++++++++ 28 files changed, 821 insertions(+), 151 deletions(-) create mode 100755 tests/fm-task-delivery.test.sh diff --git a/.agents/skills/project-management/SKILL.md b/.agents/skills/project-management/SKILL.md index f4c62a57901..8feb522bd0c 100644 --- a/.agents/skills/project-management/SKILL.md +++ b/.agents/skills/project-management/SKILL.md @@ -29,43 +29,50 @@ Apply `AGENTS.md` section 7's authoritative secondmate routing rules; if an exis Absence from the main `data/projects.md` registry is never evidence that no second mate owns the domain. If the owning second mate cannot accept the route, report that concrete blocker or obtain an explicit captain redirection rather than silently duplicating the project in the main home. -Resolve the project name, destination, delivery mode, and autonomy posture before changing local or remote state. +Resolve the project name, destination, delivery posture, and autonomy posture before changing local or remote state. Keep a newly added clone and its registry entry consistent, and roll back only artifacts created by the incomplete operation when a later initialization step fails and that rollback is safe. Do not overwrite or repurpose an existing path. ## Delivery posture -Choose the delivery mode when adding or creating the project: +The registry records the project's standing posture, which is the captain's default for the work rather than any task's answer; `AGENTS.md` section 7 owns how each task's concrete mode and yolo are resolved at intake and passed explicitly to the brief, the spawn, and any promotion. +Choose that posture when adding or creating the project: -- `no-mistakes` runs the full validation pipeline before a PR and is the default when the captain does not specify a mode. +- `no-mistakes` runs the full validation pipeline before a PR. - `direct-PR` pushes and opens a PR without the no-mistakes pipeline. - `local-only` has no required remote or PR and lands only through the approved local fast-forward path. +- `no-mistakes-prod-only` is a conditional policy rather than one flat mode: genuinely internal-only tooling, automation, contributor or operator process, and release or submission work ships `direct-PR`, while product-facing, mixed, and uncertain work ships `no-mistakes`. + +`no-mistakes-prod-only` is the default for a newly added or created remote-backed project when the captain specifies nothing, and a project with no remote defaults to `local-only`. +State that resolved default while confirming the source, local name, and posture instead of asking the captain to choose from scratch, and record a flat mode instead whenever they ask for one. +Existing registry entries keep the meaning they already have and are never migrated or reinterpreted, so a legacy entry with no bracket stays `no-mistakes`. +Registering a conditional policy is a one-time choice and never requires classifying any change; the per-task surface classification happens at each task's intake, and internal-only is never inferred from file location or project name. The optional `+yolo` posture changes routine approval authority but does not change the delivery mode. -Default it off, and enable it only on the captain's explicit instruction. +Default it off for every project and every posture, and enable it only on the captain's explicit instruction. `AGENTS.md` section 7 owns the complete authority boundary and exceptions when it is on. ## Add or clone an existing project -Confirm the source URL, local project name, delivery mode, and autonomy posture. +Confirm the source URL, local project name, delivery posture, and autonomy posture, stating the resolved default for each rather than asking the captain to invent one. Clone into `projects/` and add the registry entry only after the destination is known to be unused. -A `no-mistakes` project must have an `origin` remote and must complete the initialization procedure below. +A `no-mistakes` or `no-mistakes-prod-only` project must have an `origin` remote and must complete the initialization procedure below, because a conditional policy's product-facing work runs the pipeline while its internal-only work still takes the direct PR. A `direct-PR` project needs an `origin` remote but skips no-mistakes initialization. A `local-only` project may have no remote and skips no-mistakes initialization. ## Create a project Creating a GitHub repository is outward-facing. -Before making that remote change, propose the repository name, owner or organization, visibility, and delivery mode, defaulting visibility to private and delivery mode to `no-mistakes`, then obtain the captain's explicit consent for those values. +Before making that remote change, propose the repository name, owner or organization, visibility, and delivery posture, defaulting visibility to private and the posture to `no-mistakes-prod-only`, then obtain the captain's explicit consent for those exact values; a stated default never replaces that consent. Use `gh-axi` for the approved GitHub operation and consult its current help rather than relying on remembered flags. -After remote creation succeeds, clone it locally, add the registry entry, and initialize it according to its delivery mode. +After remote creation succeeds, clone it locally, add the registry entry, and initialize it according to its delivery posture. For a purely `local-only` project, create a local Git repository under its unused `projects/` path, add the registry entry, and make no GitHub call. The captain's request to create that local project authorizes this local initialization, but it does not authorize an unmentioned remote repository. ## Initialize -Run no-mistakes initialization only for `no-mistakes` projects: +Run no-mistakes initialization only for `no-mistakes` and `no-mistakes-prod-only` projects: ```sh cd projects/ && no-mistakes init && no-mistakes doctor diff --git a/AGENTS.md b/AGENTS.md index 6e8eea580ad..2adf7422f05 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -80,7 +80,7 @@ data/ personal fleet records; LOCAL, gitignored as a whole captain.md this home's domain-local captain preferences and working style; LOCAL, gitignored, canonical even if harness memory mirrors it, and updated with inspect-then-update captain-shared.md main-authoritative shared captain preferences propagated read-only to secondmate homes; LOCAL, gitignored, owned by secondmate-provisioning learnings.md fleet-local operational facts and gotchas; LOCAL, gitignored; dated, evidence-backed, curated, and updated with inspect-then-update - rewrite and prune rather than append forever, the same contract as captain.md; created lazily, absent until this home has a learning to store - projects.md thin fleet navigation registry; firstmate-private, parsed by fm-project-mode.sh (section 6) + projects.md thin fleet navigation registry recording each project's standing delivery posture; firstmate-private, parsed for mechanical sync and seeding by fm-project-mode.sh (section 6) secondmates.md secondmate routing table; firstmate-private, maintained by fm-home-seed.sh (section 6) /brief.md per-task crewmate brief, or per-secondmate charter brief when kind=secondmate /report.md scout task deliverable, written by the crewmate; survives teardown @@ -258,6 +258,12 @@ Never both present a likely-enough solution and launch a parallel design exercis A diagnostic request, report, recommendation, or implementation-ready finding is evidence, not authorization to change code. Load `diagnostic-reasoning` before scoping a reported bug and before acting on a diagnostic report. +Resolve every ship task's concrete delivery mode and yolo posture at intake, and pass both explicitly to the brief, the spawn, and any scout promotion, which all refuse to guess. +A current explicit captain instruction wins; otherwise the project's registry entry is the captain's standing posture, and dropping below its rigor needs a reason you can state. +On a `no-mistakes-prod-only` project, classify the task's surface: internal-only tooling, automation, contributor or operator process, and release or submission work ships `direct-PR`, while product-facing, mixed, and uncertain work ships `no-mistakes`; never infer internal-only from file location or project name. +An unregistered project or absent registry resolves to `no-mistakes` with yolo off, and the registration gap goes to the captain. +Record the resulting mode, yolo, and the one-line reason for any deviation in the backlog item note. + Treat file or subsystem overlap as a risk signal rather than an automatic reason to wait, and dispatch isolated work immediately with no concurrency cap when each change can be independently implemented and validated and the selected delivery path can reconcile ordinary rebases or conflicts. Serialize only for a true semantic dependency, shared mutable external state, incompatible concurrent migration, or another concrete condition that makes independent progress or reconciliation unsafe; same-file editing alone is insufficient, and genuine blockers remain durable. Write the task-specific brief under section 11 before spawning. diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index fc289dd6e86..79f835e342d 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -6,7 +6,8 @@ # description, acceptance criteria, and context, and may adjust other sections # when the task genuinely deviates (e.g. working an existing external PR instead # of shipping a new one). -# Usage: fm-brief.sh [--scout] [--herdr-lab] +# Usage: fm-brief.sh --mode [--herdr-lab] +# fm-brief.sh --scout [--herdr-lab] # fm-brief.sh --secondmate {...|--no-projects} # --scout writes the scout contract instead: the deliverable is a report at # data//report.md (no branch, no push, no PR) and the worktree is scratch. @@ -26,15 +27,24 @@ # The flag must be explicit because {TASK} is filled after scaffolding and the # caller-supplied repo string cannot reliably identify this repo. Briefs made # without it carry a loud declaration so an omitted contract cannot be silent. -# For ship tasks, the definition of done is shaped by the project's delivery mode -# (data/projects.md via fm-project-mode.sh; see the project-management skill -# and AGENTS.md task lifecycle): -# no-mistakes implement -> /no-mistakes pipeline -> PR -> captain merge (default) -# direct-PR implement -> push + open PR via gh-axi (no pipeline) -> captain merge +# For ship tasks, --mode is REQUIRED and shapes the definition of done. Firstmate +# resolves it per task at intake (AGENTS.md section 7); data/projects.md holds the +# captain's standing posture as context, and this script never reads it: +# no-mistakes implement -> /no-mistakes pipeline -> PR -> configured merge authority +# direct-PR implement -> push + open PR via gh-axi (no pipeline) -> configured merge authority # local-only implement on branch, stop and report "ready in branch" (no push/PR); -# captain approves, firstmate merges to local main +# the configured merge authority approves, firstmate merges to local main +# no-mistakes-prod-only is a registry policy, not a task mode; resolve it to one of +# the three concrete modes at intake before calling this script. +# The generated ship brief records the chosen mode as a fixed machine-readable +# "Delivery contract: mode=" line. bin/fm-spawn.sh reads that line and refuses +# to launch a ship task whose explicit --mode disagrees, so an adjusted brief and the +# recorded task metadata cannot drift apart. # Ship briefs begin with a worktree-isolation assertion before the branch step. -# Scout tasks ignore mode - their deliverable is a report, not a merge. +# --mode is refused on scout and secondmate scaffolds: a scout's deliverable is a +# report rather than a merge, and a charter is not a delivery contract. +# There is no --yolo flag here. The worker never owns approval decisions, so yolo is +# a spawn-time and firstmate-side input only (AGENTS.md section 7). # Every scaffold's status protocol distinguishes the configured # declared-external-wait verb (FM_CLASSIFY_PAUSED_VERB, default "paused") from # "blocked:": pause for a known external wait expected to clear on its own, @@ -94,16 +104,56 @@ fi KIND=ship HERDR_LAB=0 NO_PROJECTS=0 +MODE= +MODE_SET=0 POS=() +want_value= for a in "$@"; do + if [ -n "$want_value" ]; then + case "$a" in + --*) echo "error: --$want_value requires a value" >&2; exit 1 ;; + esac + case "$want_value" in + mode) MODE=$a; MODE_SET=1 ;; + *) echo "error: internal parser state for --$want_value" >&2; exit 1 ;; + esac + want_value= + continue + fi case "$a" in --scout) KIND=scout ;; --secondmate) KIND=secondmate ;; --herdr-lab) HERDR_LAB=1 ;; --no-projects) NO_PROJECTS=1 ;; + --mode) want_value=mode ;; + --mode=*) MODE=${a#--mode=}; MODE_SET=1 ;; + # yolo never reaches the worker: it is firstmate's approval authority, not a + # brief input. Refuse it loudly so it is never silently dropped here and then + # believed to have been recorded. + --yolo|--yolo=*) echo "error: --yolo is not a brief input; pass it to bin/fm-spawn.sh, which records the task's approval posture" >&2; exit 1 ;; *) POS+=("$a") ;; esac done +[ -z "$want_value" ] || { echo "error: --$want_value requires a value" >&2; exit 1; } + +# Ship delivery mode is an explicit per-task decision (AGENTS.md section 7). A +# missing or invalid value stops the scaffold rather than silently defaulting. +if [ "$KIND" = ship ]; then + [ "$MODE_SET" -eq 1 ] || { + echo "error: ship briefs require --mode ; resolve it at intake from the captain's instruction and the project's registered posture in data/projects.md" >&2 + exit 1 + } + case "$MODE" in + no-mistakes|direct-PR|local-only) ;; + no-mistakes-prod-only) + echo "error: no-mistakes-prod-only is a registry policy, not a task mode; classify this task's surface and resolve it to no-mistakes or direct-PR at intake" >&2 + exit 1 ;; + *) echo "error: --mode must be one of no-mistakes, direct-PR, local-only (got '$MODE')" >&2; exit 1 ;; + esac +elif [ "$MODE_SET" -eq 1 ]; then + echo "error: --mode applies only to ship briefs; a scout delivers a report and a secondmate charter is not a delivery contract" >&2 + exit 1 +fi ID=${POS[0]} if [ "$KIND" = secondmate ] && [ "$HERDR_LAB" -eq 1 ]; then @@ -295,20 +345,18 @@ echo "scaffolded: $BRIEF (scout; replace {TASK})" exit 0 fi -# Ship task: shape Setup / Rule 1 / Definition of done by the project's delivery mode. -# yolo does not affect the brief because the worker never owns approval decisions; -# firstmate applies the authority contract in AGENTS.md section 7, so discard it. -read -r MODE _ <" line that bin/fm-spawn.sh checks against its own +# explicit --mode before launching. case "$MODE" in direct-PR) SETUP2="" RULE1='1. Never push to the default branch (push only your `fm/'"$ID"'` branch). Never merge a PR.' IFS= read -r -d '' DOD < " where mode is one of # no-mistakes|direct-PR|local-only and yolo is on|off. # +# MECHANICAL CONSUMERS ONLY. This answers "what posture did the captain register +# for this project", never "how does this task ship". A task's delivery mode and +# yolo are resolved by firstmate at intake and passed explicitly to +# bin/fm-brief.sh, bin/fm-spawn.sh, and bin/fm-promote.sh (AGENTS.md section 7). +# The consumers are bin/fm-fleet-sync.sh (skip local-only clones), +# bin/fm-home-seed.sh (refuse local-only seeding, run no-mistakes init), and +# bin/fm-spawn.sh's advisory registry-deviation notice. +# # Registry line format (data/projects.md): # - - (added ) -> no-mistakes off (legacy default) # - [] - (added ) -> off # - [ +yolo] - (added ) -> on # -# mode = how a finished change reaches main: -# no-mistakes full pipeline -> PR -> captain merge (default) -# direct-PR push + PR via gh-axi, no pipeline -> captain merge -# local-only local branch, no remote/PR -> captain approve -> guarded local merge +# Registered modes: +# no-mistakes full pipeline -> PR -> configured merge authority (default) +# direct-PR push + PR via gh-axi, no pipeline +# local-only local branch, no remote/PR, guarded local merge +# no-mistakes-prod-only a conditional policy, not a task mode: firstmate +# classifies each task's surface at intake (the +# project-management skill owns that classification). +# Mechanical output maps it to its most rigorous leg, +# no-mistakes, so sync, seeding, and init treat such a +# project as the remote-backed pipeline project it is. # yolo (orthogonal) = when on, firstmate may make routine approval decisions itself. # AGENTS.md section 7 is the single owner of authority exceptions, including # ask-user contract expansion and stronger captain boundaries. # +# --raw prints the registered annotation unmapped, so a caller that must tell a +# conditional policy apart from a flat mode sees "no-mistakes-prod-only" itself. +# # An unknown/missing project or unknown mode falls back to "no-mistakes off" and warns # to stderr, so a typo never silently drops the gate. -# Usage: fm-project-mode.sh +# Usage: fm-project-mode.sh [--raw] set -eu SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -26,7 +43,12 @@ FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" DATA="${FM_DATA_OVERRIDE:-$FM_HOME/data}" REG="$DATA/projects.md" -NAME=${1:?usage: fm-project-mode.sh } +RAW=0 +if [ "${1:-}" = "--raw" ]; then + RAW=1 + shift +fi +NAME=${1:?usage: fm-project-mode.sh [--raw] } if [ ! -f "$REG" ]; then echo "warn: no registry at $REG; defaulting $NAME to no-mistakes off" >&2 @@ -59,8 +81,13 @@ fi mode=${parsed%% *} yolo=${parsed##* } case "$mode" in - no-mistakes|direct-PR|local-only) ;; + no-mistakes|direct-PR|local-only|no-mistakes-prod-only) ;; *) echo "warn: unknown mode \"$mode\" for $NAME; defaulting to no-mistakes off" >&2; mode=no-mistakes; yolo=off ;; esac case "$yolo" in on|off) ;; *) yolo=off ;; esac +# A conditional policy is not a task mode. Mechanical callers get its most +# rigorous leg; --raw callers get the annotation itself (see the header). +if [ "$RAW" -eq 0 ] && [ "$mode" = no-mistakes-prod-only ]; then + mode=no-mistakes +fi echo "$mode $yolo" diff --git a/bin/fm-promote.sh b/bin/fm-promote.sh index 827c17998f2..92c3e53448a 100755 --- a/bin/fm-promote.sh +++ b/bin/fm-promote.sh @@ -5,25 +5,84 @@ # again. After promoting, send the crewmate its ship instructions via fm-send.sh # (inventory scratch state, reset to a clean default-branch base, carry over only # intended fix changes, create branch fm/, implement, then report done -# according to the project's delivery mode). -# Usage: fm-promote.sh +# according to this task's delivery mode). +# A scout records no delivery posture, so promotion is where this task's delivery +# contract is decided: --mode and --yolo are REQUIRED and written into the meta +# alongside the kind= flip. Firstmate resolves both at promotion time, having just +# read the scout's report (AGENTS.md section 7); data/projects.md holds the +# captain's standing posture as context, and this script never looks it up. +# no-mistakes-prod-only is a registry policy rather than a task mode and is refused. +# Usage: fm-promote.sh --mode --yolo set -eu SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" + +MODE= +YOLO= +MODE_SET=0 +YOLO_SET=0 +POS=() +want_value= +for a in "$@"; do + if [ -n "$want_value" ]; then + case "$a" in + --*) echo "error: --$want_value requires a value" >&2; exit 1 ;; + esac + case "$want_value" in + mode) MODE=$a; MODE_SET=1 ;; + yolo) YOLO=$a; YOLO_SET=1 ;; + esac + want_value= + continue + fi + case "$a" in + --mode) want_value=mode ;; + --mode=*) MODE=${a#--mode=}; MODE_SET=1 ;; + --yolo) want_value=yolo ;; + --yolo=*) YOLO=${a#--yolo=}; YOLO_SET=1 ;; + *) POS+=("$a") ;; + esac +done +[ -z "$want_value" ] || { echo "error: --$want_value requires a value" >&2; exit 1; } +[ "${#POS[@]}" -ge 1 ] || { echo "usage: fm-promote.sh --mode --yolo " >&2; exit 1; } +[ "$MODE_SET" -eq 1 ] || { + echo "error: promotion requires --mode ; decide it now from the scout's findings and the project's registered posture in data/projects.md" >&2 + exit 1 +} +[ "$YOLO_SET" -eq 1 ] || { + echo "error: promotion requires --yolo ; it is this task's routine approval authority, not a project lookup" >&2 + exit 1 +} +case "$MODE" in + no-mistakes|direct-PR|local-only) ;; + no-mistakes-prod-only) + echo "error: no-mistakes-prod-only is a registry policy, not a task mode; classify this task's surface and resolve it to no-mistakes or direct-PR" >&2 + exit 1 ;; + *) echo "error: --mode must be one of no-mistakes, direct-PR, local-only (got '$MODE')" >&2; exit 1 ;; +esac +case "$YOLO" in + on|off) ;; + *) echo "error: --yolo must be on or off (got '$YOLO')" >&2; exit 1 ;; +esac + "$FM_ROOT/bin/fm-guard.sh" || true -ID=$1 +ID=${POS[0]} META="$STATE/$ID.meta" [ -f "$META" ] || { echo "error: no meta for task $ID at $META" >&2; exit 1; } grep -qx 'kind=scout' "$META" || { echo "error: task $ID is not a scout task (kind=scout not in meta)" >&2; exit 1; } TMP="$META.tmp" -grep -v '^kind=' "$META" > "$TMP" -echo "kind=ship" >> "$TMP" +grep -v -e '^kind=' -e '^mode=' -e '^yolo=' "$META" > "$TMP" +{ + echo "kind=ship" + echo "mode=$MODE" + echo "yolo=$YOLO" +} >> "$TMP" mv "$TMP" "$META" HOME_Q=$(printf '%q' "$FM_HOME") -echo "promoted $ID to ship (teardown protection restored)" -echo "next: FM_HOME=$HOME_Q bin/fm-send.sh fm-$ID ''" +echo "promoted $ID to ship mode=$MODE yolo=$YOLO (teardown protection restored)" +echo "next: FM_HOME=$HOME_Q bin/fm-send.sh fm-$ID ''" diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index b6ffdfd347d..21dba20477e 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -1,8 +1,21 @@ #!/usr/bin/env bash # Spawn a direct report: a crewmate in a treehouse or Orca worktree, or a # secondmate in its isolated firstmate home. -# Usage: fm-spawn.sh [--harness |harness|launch-command] [--model ] [--effort ] [--backend ] [--scout] +# Usage: fm-spawn.sh --mode --yolo [--harness |harness|launch-command] [--model ] [--effort ] [--backend ] +# fm-spawn.sh --scout [--harness |harness|launch-command] [--model ] [--effort ] [--backend ] # fm-spawn.sh [] [--harness |harness|launch-command] [--model ] [--effort ] [--backend ] --secondmate +# --mode and --yolo are this task's delivery contract, REQUIRED for every ship +# spawn and refused on --scout and --secondmate spawns. Firstmate resolves both +# per task at intake (AGENTS.md section 7); data/projects.md holds the captain's +# standing posture as context, not as this task's answer, so a spawn never looks +# the mode up. A ship spawn additionally reads the brief's recorded +# "Delivery contract: mode=" line and REFUSES a mismatch, so the worker's +# instructions and the recorded task delivery cannot drift apart; a brief +# scaffolded before that line existed warns once and launches on the flag. When +# the explicit mode carries less rigor than the project's standing posture, a +# loud one-line deviation notice is printed and the spawn continues. +# no-mistakes-prod-only is a registry policy rather than a task mode and is +# refused as a flag value. # --harness is the explicit per-spawn harness/profile adapter. The old # positional harness arg still works for back-compat. # --model and --effort are concrete profile @@ -99,7 +112,9 @@ # Batch dispatch: pass one or more `id=repo` pairs instead of a single , e.g. # fm-spawn.sh fix-a-k3=projects/foo add-b-q7=projects/bar [--scout] # Each pair re-execs this script in single-task mode, so the single path stays the only -# source of truth; shared --scout/--harness/--model/--effort/--backend applies to every pair. +# source of truth; shared --scout/--harness/--model/--effort/--backend/--mode/--yolo +# applies to every pair. A ship batch therefore carries one delivery contract, and each +# pair still checks it against its own brief; a batch spanning modes is two invocations. # If config/crew-dispatch.json exists, shared --harness is required for crewmate # and scout batches. The loop lives here, in bash, so callers never hand-write a # multi-task shell loop (the tool shell is zsh, which does not word-split unquoted @@ -118,9 +133,10 @@ # a firstmate-owned global hook and registry, and a gitignored per-task pointer. # grok uses a firstmate-owned global hook under ${GROK_HOME:-$HOME/.grok}/hooks # plus a gitignored .fm-grok-turnend worktree pointer and a state token. -# On success prints: spawned harness= kind= mode= yolo= window= worktree= -# mode/yolo are resolved per-project from data/projects.md for ship/scout tasks; -# secondmate spawns record mode=secondmate, yolo=off, home=, and projects=. +# On success prints: spawned harness= kind= [mode= yolo=] window= worktree= +# A ship task records the explicit mode/yolo it was passed; a secondmate spawn records +# mode=secondmate, yolo=off, home=, and projects=; a scout records neither, and both the +# success line and state/.meta omit them. set -eu SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -188,10 +204,14 @@ HARNESS_ARG= MODEL= EFFORT= BACKEND_ARG= +MODE= +YOLO= HARNESS_SET=0 MODEL_SET=0 EFFORT_SET=0 BACKEND_SET=0 +MODE_SET=0 +YOLO_SET=0 POS=() want_value= for a in "$@"; do @@ -204,6 +224,8 @@ for a in "$@"; do model) MODEL=$a; MODEL_SET=1 ;; effort) EFFORT=$a; EFFORT_SET=1 ;; backend) BACKEND_ARG=$a; BACKEND_SET=1 ;; + mode) MODE=$a; MODE_SET=1 ;; + yolo) YOLO=$a; YOLO_SET=1 ;; *) echo "error: internal parser state for --$want_value" >&2; exit 1 ;; esac want_value= @@ -220,6 +242,10 @@ for a in "$@"; do --effort=*) EFFORT=${a#--effort=}; EFFORT_SET=1 ;; --backend) want_value=backend ;; --backend=*) BACKEND_ARG=${a#--backend=}; BACKEND_SET=1 ;; + --mode) want_value=mode ;; + --mode=*) MODE=${a#--mode=}; MODE_SET=1 ;; + --yolo) want_value=yolo ;; + --yolo=*) YOLO=${a#--yolo=}; YOLO_SET=1 ;; *) POS+=("$a") ;; esac done @@ -228,11 +254,48 @@ done [ "$MODEL_SET" -eq 0 ] || [ -n "$MODEL" ] || { echo "error: --model requires a non-empty value" >&2; exit 1; } [ "$EFFORT_SET" -eq 0 ] || [ -n "$EFFORT" ] || { echo "error: --effort requires a non-empty value" >&2; exit 1; } [ "$BACKEND_SET" -eq 0 ] || [ -n "$BACKEND_ARG" ] || { echo "error: --backend requires a non-empty value" >&2; exit 1; } +[ "$MODE_SET" -eq 0 ] || [ -n "$MODE" ] || { echo "error: --mode requires a non-empty value" >&2; exit 1; } +[ "$YOLO_SET" -eq 0 ] || [ -n "$YOLO" ] || { echo "error: --yolo requires a non-empty value" >&2; exit 1; } case "$EFFORT" in ''|low|medium|high|xhigh|max) ;; *) echo "error: --effort must be one of low, medium, high, xhigh, max" >&2; exit 1 ;; esac +# Delivery contract (AGENTS.md section 7). A ship task's mode and yolo are +# firstmate's per-task decision, so they are required and closed-set validated +# here rather than resolved from the project registry. Scouts deliver a report +# and record no delivery posture; secondmate spawns hardcode theirs. +if [ "$KIND" = ship ]; then + [ "$MODE_SET" -eq 1 ] || { + echo "error: ship spawns require --mode ; resolve it at intake from the captain's instruction and the project's registered posture in data/projects.md" >&2 + exit 1 + } + [ "$YOLO_SET" -eq 1 ] || { + echo "error: ship spawns require --yolo ; it is this task's routine approval authority, not a project lookup" >&2 + exit 1 + } + case "$MODE" in + no-mistakes|direct-PR|local-only) ;; + no-mistakes-prod-only) + echo "error: no-mistakes-prod-only is a registry policy, not a task mode; classify this task's surface and resolve it to no-mistakes or direct-PR at intake" >&2 + exit 1 ;; + *) echo "error: --mode must be one of no-mistakes, direct-PR, local-only (got '$MODE')" >&2; exit 1 ;; + esac + case "$YOLO" in + on|off) ;; + *) echo "error: --yolo must be on or off (got '$YOLO')" >&2; exit 1 ;; + esac +else + [ "$MODE_SET" -eq 0 ] || { + echo "error: --mode applies only to ship spawns; a scout delivers a report and a secondmate records its own fixed posture" >&2 + exit 1 + } + [ "$YOLO_SET" -eq 0 ] || { + echo "error: --yolo applies only to ship spawns; a scout delivers a report and a secondmate records its own fixed posture" >&2 + exit 1 + } +fi + # Backend selection (data/fm-backend-design-d7): explicit --backend, else # FM_BACKEND env, else config/backend, else runtime auto-detection, else # default tmux (fm_backend_name). fm_backend_validate_spawn refuses unknown or @@ -324,8 +387,8 @@ spawn_abort_cleanup() { echo "project=$PROJ_ABS" echo "harness=$HARNESS" echo "kind=$KIND" - echo "mode=${MODE:-no-mistakes}" - echo "yolo=${YOLO:-off}" + [ -z "${MODE:-}" ] || echo "mode=$MODE" + [ -z "${YOLO:-}" ] || echo "yolo=$YOLO" echo "tasktmp=${TASK_TMP:-}" echo "model=${MODEL:-default}" echo "effort=${EFFORT:-default}" @@ -394,6 +457,11 @@ if [ "${#POS[@]}" -gt 0 ] && [ "${POS[0]}" != "$idpart" ] && case "$idpart" in * [ -z "$MODEL" ] || shared_args+=(--model "$MODEL") [ -z "$EFFORT" ] || shared_args+=(--effort "$EFFORT") [ -z "$BACKEND_ARG" ] || shared_args+=(--backend "$BACKEND_ARG") + # One delivery contract applies to every pair in a batch, exactly like the shared + # harness. Each pair still re-validates it against its own brief, so a batch + # spanning several modes is two invocations rather than a silent mixed dispatch. + [ "$MODE_SET" -eq 0 ] || shared_args+=(--mode "$MODE") + [ "$YOLO_SET" -eq 0 ] || shared_args+=(--yolo "$YOLO") for pair in "${POS[@]}"; do case "$pair" in *=*) : ;; @@ -845,6 +913,41 @@ else BRIEF="$DATA/$ID/brief.md" fi [ -f "$BRIEF" ] || { echo "error: no brief at $BRIEF" >&2; exit 1; } + +delivery_rigor_rank() { # -> 3 (most rigor) .. 1 (least); 0 = not a task mode + case "$1" in + no-mistakes) echo 3 ;; + direct-PR) echo 2 ;; + local-only) echo 1 ;; + *) echo 0 ;; + esac +} + +# Brief/spawn delivery agreement, checked before any endpoint exists. +# fm-brief.sh records a ship brief's mode as a fixed "Delivery contract: mode=" +# line. A spawn that disagrees would launch a worker whose instructions and whose +# recorded task delivery differ, which is the exact drift this contract prevents. +if [ "$KIND" = ship ]; then + PROJ_NAME=$(basename "$PROJ_ABS") + BRIEF_MODE=$(sed -n 's/^Delivery contract: mode=\([^ ]*\).*$/\1/p' "$BRIEF" | head -n 1) + if [ -z "$BRIEF_MODE" ]; then + echo "warning: $BRIEF records no delivery contract line (scaffolded before ship briefs recorded one); launching on the explicit --mode $MODE - confirm its definition of done matches" >&2 + elif [ "$BRIEF_MODE" != "$MODE" ]; then + echo "error: delivery mismatch for $ID: the brief says mode=$BRIEF_MODE but this spawn passed --mode $MODE; correct the flag or re-scaffold the brief so the worker's instructions and the task record agree" >&2 + exit 1 + fi + # The registry holds the captain's standing posture, so dropping below it is + # allowed (a current explicit captain instruction wins) but never silent. An + # unregistered project resolves to the same no-mistakes standing default, which + # is why the notice names the standing posture rather than the registry line. A + # conditional policy is excluded: both of its legs are legitimate classifications. + STANDING_MODE=$("$FM_ROOT/bin/fm-project-mode.sh" --raw "$PROJ_NAME" 2>/dev/null | cut -d' ' -f1) || STANDING_MODE= + if [ -n "$STANDING_MODE" ] && [ "$STANDING_MODE" != no-mistakes-prod-only ] \ + && [ "$(delivery_rigor_rank "$MODE")" -lt "$(delivery_rigor_rank "$STANDING_MODE")" ]; then + echo "notice: $ID ships mode=$MODE while the standing posture for $PROJ_NAME is $STANDING_MODE - less rigor than the captain's standing posture; proceed only on a current explicit captain instruction or an intake judgment you can state" >&2 + fi +fi + BRIEF_DIR_REAL=$(cd "$(dirname "$BRIEF")" && pwd -P) BRIEF_REAL="$BRIEF_DIR_REAL/$(basename "$BRIEF")" @@ -1585,19 +1688,19 @@ EOF esac fi -# Per-project delivery mode + yolo flag (bin/fm-project-mode.sh; the project-management skill and AGENTS.md task lifecycle). -# Recorded in meta so fm-teardown's safety check and the validate/merge stages can -# branch on them. Mode governs ship tasks; a scout's deliverable is a report, not a -# merge, so scout teardown ignores mode. +# Delivery posture recorded in meta so fm-teardown's safety check and the +# validate/merge stages can branch on it. A ship task carries the explicit +# per-task decision validated above; a secondmate's posture is fixed; a scout +# records none at all, because its deliverable is a report rather than a merge +# (fm-teardown.sh defaults an absent mode to no-mistakes, and fm-promote.sh +# requires an explicit mode when a scout is promoted to a ship task). if [ "$KIND" = secondmate ]; then MODE=secondmate YOLO=off : "${SECONDMATE_PROJECTS:=}" -else - PROJ_NAME=$(basename "$PROJ_ABS") - read -r MODE YOLO </head` by default (recorded `pr_head=` is only an offline fallback) before falling back to the local branch with a warning. For target project repos shipped through their own no-mistakes pipeline, commits under `.no-mistakes/evidence/` are the pipeline's PR-viewable validation evidence and are expected to stay in the crew branch until the evidence-hosting design changes. The firstmate repo itself is the exception: its `.no-mistakes/` directory is local state, stays gitignored, and is rejected by CI if tracked. diff --git a/docs/scripts.md b/docs/scripts.md index 767980b75af..358734d6fc8 100644 --- a/docs/scripts.md +++ b/docs/scripts.md @@ -18,7 +18,7 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | `fm-update.sh` | Fast-forward-only self-update of firstmate and secondmate homes from origin | | `fm-backlog-handoff.sh` | Validate and delegate queued backlog-item moves into a secondmate home | | `fm-decision-hold.sh` | Create, verify, complete, and resolve durable captain-held decisions | -| `fm-brief.sh` | Scaffold ship, scout, secondmate-charter, and Herdr-lab briefs | +| `fm-brief.sh` | Scaffold ship (explicit `--mode`), scout, secondmate-charter, and Herdr-lab briefs | | `fm-herdr-lab.sh` | Provision and guardedly operate an isolated, never-default Herdr lab session | | `fm-install-herdr.sh` | Install CI's exact-version Herdr pin with official asset URL, SHA-256, and protocol checks | | `fm-install-treehouse.sh`| Install CI's exact-version Treehouse pin for real-Herdr E2E that needs spawn worktrees | @@ -48,7 +48,7 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | `backends/orca.sh` | Experimental Orca backend adapter owning both worktree and terminal | | `backends/cmux.sh` | Experimental cmux session-provider adapter | | `fm-config-push.sh` | Push declared inherited local material to live secondmates mid-session and send a pointer to the literal-content config reread when config changed | -| `fm-project-mode.sh` | Resolve a project's delivery mode and `+yolo` flag from `data/projects.md` | +| `fm-project-mode.sh` | Resolve a project's registered delivery posture from `data/projects.md` for fleet sync and home seeding | | `fm-merge-local.sh` | Fast-forward a `local-only` project's local default branch after approval | | `fm-review-diff.sh` | Review a crewmate branch or resolved PR head against the authoritative base | | `fm-marker-lib.sh` | Compatibility entry point for the from-firstmate carrier owned by `fm-operational-input.sh` | @@ -87,7 +87,7 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | `fm-pr-check-migrate.sh` | Quarantine older task polls without execution and rebuild only canonical polls | | `fm-pr-check.sh` | Record validated `pr=` and `pr_head=` values, then atomically arm a static merge poll | | `fm-pr-merge.sh` | Record PR metadata, then merge a task's canonical full GitHub URL | -| `fm-promote.sh` | Promote a scout task in place to a protected ship task | +| `fm-promote.sh` | Promote a scout task in place to a protected ship task with an explicit delivery mode | | `fm-teardown.sh` | Fail-closed teardown: return landed ship worktrees, require completed scout deliverables, retire secondmate homes | | `fm-harness.sh` | Detect the running harness and resolve crew or secondmate harness, model, and effort | | `fm-lock.sh` | Per-home firstmate session lock | diff --git a/tests/fm-ask-user-authority.test.sh b/tests/fm-ask-user-authority.test.sh index 89ec517fa1b..469eb92c2a2 100644 --- a/tests/fm-ask-user-authority.test.sh +++ b/tests/fm-ask-user-authority.test.sh @@ -14,7 +14,7 @@ test_primary_and_secondmate_instruction_generation() { mkdir -p "$home/data" FM_HOME="$home" FM_ROOT_OVERRIDE="$ROOT" \ - "$BRIEF" authority-worker sample >/dev/null 2>&1 + "$BRIEF" authority-worker sample --mode no-mistakes >/dev/null 2>&1 ship="$home/data/authority-worker/brief.md" assert_grep 'ask-user findings are never yours to answer' "$ship" \ "generated implementation brief lets the worker own an ask-user decision" diff --git a/tests/fm-backend-autodetect-smoke.test.sh b/tests/fm-backend-autodetect-smoke.test.sh index 17fe88f6171..54bdf854849 100755 --- a/tests/fm-backend-autodetect-smoke.test.sh +++ b/tests/fm-backend-autodetect-smoke.test.sh @@ -103,7 +103,7 @@ env -u TMUX -u FM_BACKEND PATH="$PATH" HERDR_ENV=1 \ FM_ROOT_OVERRIDE="$ROOT" FM_STATE_OVERRIDE="$STATE" FM_DATA_OVERRIDE="$DATA" \ FM_CONFIG_OVERRIDE="$CONFIG" FM_PROJECTS_OVERRIDE="$TMP_ROOT/unused-projects" \ FM_SPAWN_NO_GUARD=1 \ - "$ROOT/bin/fm-spawn.sh" "$ID" "$PROJ" "sh -c 'echo autodetect-smoke-ok'" \ + "$ROOT/bin/fm-spawn.sh" "$ID" "$PROJ" "sh -c 'echo autodetect-smoke-ok'" --mode no-mistakes --yolo off \ >"$OUT_FILE" 2>"$ERR_FILE" status=$? [ "$status" -eq 0 ] || fail "fm-spawn.sh did not succeed auto-detecting herdr"$'\n'"--- stdout ---"$'\n'"$(cat "$OUT_FILE")"$'\n'"--- stderr ---"$'\n'"$(cat "$ERR_FILE")" diff --git a/tests/fm-backend-herdr-launcher-workspace-e2e.test.sh b/tests/fm-backend-herdr-launcher-workspace-e2e.test.sh index ca5cc4575e9..3cb8b49d0dc 100755 --- a/tests/fm-backend-herdr-launcher-workspace-e2e.test.sh +++ b/tests/fm-backend-herdr-launcher-workspace-e2e.test.sh @@ -201,7 +201,7 @@ focused_workspace() { # --- 1. unique label, no herdr ancestry: the per-home container still works -- -spawn_from_launcher "" "$PRIMARY_HOME" uniqA "$PROJ" +spawn_from_launcher "" "$PRIMARY_HOME" uniqA "$PROJ" --mode no-mistakes --yolo off [ "$SPAWN_RC" -eq 0 ] || fail "a primary-shaped spawn with no herdr parent failed"$'\n'"$(cat "$SPAWN_ERR")" UNIQA_META="$PRIMARY_HOME/state/uniqA.meta" record_worktree "$UNIQA_META" @@ -221,7 +221,7 @@ $(lab tab create --workspace "$WS_PRIMARY" --cwd "$TMP_ROOT" --label captain-she EOF [ -n "$LAUNCH_PRIMARY_PANE" ] || fail "could not create a launcher pane inside the 'firstmate' workspace" -spawn_from_launcher "$LAUNCH_PRIMARY_PANE" "$PRIMARY_HOME" uniqB "$PROJ" +spawn_from_launcher "$LAUNCH_PRIMARY_PANE" "$PRIMARY_HOME" uniqB "$PROJ" --mode no-mistakes --yolo off [ "$SPAWN_RC" -eq 0 ] || fail "a primary spawn from a launcher pane failed"$'\n'"$(cat "$SPAWN_ERR")" UNIQB_META="$PRIMARY_HOME/state/uniqB.meta" record_worktree "$UNIQB_META" @@ -233,7 +233,7 @@ pass "real herdr E2E: the normal unique-label path is unchanged when the launche # --- 2b. presentation spaces ON: the projected child is created and bound # UNDER the launcher's exact workspace, not collapsed into it --------- -spawn_from_launcher "$LAUNCH_PRIMARY_PANE" "$PRES_HOME" presU "$PROJ" +spawn_from_launcher "$LAUNCH_PRIMARY_PANE" "$PRES_HOME" presU "$PROJ" --mode no-mistakes --yolo off [ "$SPAWN_RC" -eq 0 ] || fail "a presentation-enabled spawn from a launcher pane failed"$'\n'"$(cat "$SPAWN_ERR")" PRESU_META="$PRES_HOME/state/presU.meta" record_worktree "$PRESU_META" @@ -273,7 +273,7 @@ cat > "$TMP_ROOT/spawn-in-pane.sh" < "$TMP_ROOT/dupC.out" 2> "$TMP_ROOT/dupC.err" echo \$? > "$TMP_ROOT/dupC.rc" SPAWN @@ -308,7 +308,7 @@ pass "real herdr E2E: the duplicate-labeled sibling workspace is left entirely u # --- 3b. presentation spaces ON with a duplicated parent label: the projection # still hangs off the launcher's exact workspace --------------------- -spawn_from_launcher "$LAUNCH_DUP_PANE" "$PRES_HOME" presD "$PROJ" +spawn_from_launcher "$LAUNCH_DUP_PANE" "$PRES_HOME" presD "$PROJ" --mode no-mistakes --yolo off [ "$SPAWN_RC" -eq 0 ] || fail "a projected spawn under a duplicated parent label failed"$'\n'"$(cat "$SPAWN_ERR")" PRESD_META="$PRES_HOME/state/presD.meta" record_worktree "$PRESD_META" @@ -335,7 +335,7 @@ pass "real herdr E2E: with a duplicated home label, a projected worker still han # --- 4. duplicate label with NO launcher identity refuses before publishing -- -spawn_from_launcher "" "$PRIMARY_HOME" dupD "$PROJ" +spawn_from_launcher "" "$PRIMARY_HOME" dupD "$PROJ" --mode no-mistakes --yolo off [ "$SPAWN_RC" -ne 0 ] || fail "a duplicate-labeled home workspace with no herdr parent must refuse, not guess" assert_contains_local "$(cat "$SPAWN_ERR")" "labeled 'firstmate'" \ "the refusal did not name the duplicated home label" @@ -359,7 +359,7 @@ if lab pane get "$STALE_PANE" >/dev/null 2>&1; then fail "the launcher pane did not actually go away" fi -spawn_from_launcher "$STALE_PANE" "$PRIMARY_HOME" staleF "$PROJ" +spawn_from_launcher "$STALE_PANE" "$PRIMARY_HOME" staleF "$PROJ" --mode no-mistakes --yolo off [ "$SPAWN_RC" -ne 0 ] || fail "a launcher pane that no longer exists must refuse, not fall back to a label search" assert_contains_local "$(cat "$SPAWN_ERR")" "$STALE_PANE" \ "the stale-identity refusal did not name the launcher pane it could not resolve" @@ -379,7 +379,7 @@ EOF [ -n "$WS_SM_DECOY" ] && [ -n "$WS_SM_LAUNCH" ] || fail "could not create the two secondmate-labeled workspaces" WS_SM_DECOY_TABS_BEFORE=$(tab_labels_of_workspace "$WS_SM_DECOY") -spawn_from_launcher "$LAUNCH_SM_PANE" "$SM_HOME" smE "$PROJ" +spawn_from_launcher "$LAUNCH_SM_PANE" "$SM_HOME" smE "$PROJ" --mode no-mistakes --yolo off [ "$SPAWN_RC" -eq 0 ] || fail "a secondmate-owned crewmate spawn failed"$'\n'"$(cat "$SPAWN_ERR")" SME_META="$SM_HOME/state/smE.meta" record_worktree "$SME_META" diff --git a/tests/fm-backend-herdr-presentation-e2e.test.sh b/tests/fm-backend-herdr-presentation-e2e.test.sh index 3691e165934..d168f3deae2 100755 --- a/tests/fm-backend-herdr-presentation-e2e.test.sh +++ b/tests/fm-backend-herdr-presentation-e2e.test.sh @@ -383,7 +383,7 @@ make_project() { # spawn_task() { # local id=$1 home=$2 project=$3 FM_GATE_REFUSE_BYPASS=1 FM_SPAWN_NO_GUARD=1 FM_HOME="$home" FM_ROOT_OVERRIDE="$ROOT" \ - "$ROOT/bin/fm-spawn.sh" "$id" "$project" "sh -c 'sleep 120'" --backend herdr + "$ROOT/bin/fm-spawn.sh" "$id" "$project" "sh -c 'sleep 120'" --mode no-mistakes --yolo off --backend herdr } spawn_secondmate_task() { diff --git a/tests/fm-backend-herdr-workspace-per-home-e2e.test.sh b/tests/fm-backend-herdr-workspace-per-home-e2e.test.sh index 110017e9b84..1cb2f1f5a87 100755 --- a/tests/fm-backend-herdr-workspace-per-home-e2e.test.sh +++ b/tests/fm-backend-herdr-workspace-per-home-e2e.test.sh @@ -111,7 +111,7 @@ PROJ2="$TMP_ROOT/scratch-project-2"; make_scratch_project "$PROJ2" CM1_OUT="$TMP_ROOT/cm1.out"; CM1_ERR="$TMP_ROOT/cm1.err" FM_SPAWN_NO_GUARD=1 FM_HOME="$PRIMARY_HOME" FM_ROOT_OVERRIDE="$ROOT" \ - "$ROOT/bin/fm-spawn.sh" cm1 "$PROJ1" "sh -c 'echo primary-crew-ok'" --backend herdr \ + "$ROOT/bin/fm-spawn.sh" cm1 "$PROJ1" "sh -c 'echo primary-crew-ok'" --mode no-mistakes --yolo off --backend herdr \ >"$CM1_OUT" 2>"$CM1_ERR" rc=$? [ "$rc" -eq 0 ] || fail "primary-shaped crewmate spawn failed"$'\n'"--- stdout ---"$'\n'"$(cat "$CM1_OUT")"$'\n'"--- stderr ---"$'\n'"$(cat "$CM1_ERR")" @@ -166,7 +166,7 @@ pass "real herdr E2E: a --secondmate spawn by the PRIMARY lands in the SECONDMAT CM2_OUT="$TMP_ROOT/cm2.out"; CM2_ERR="$TMP_ROOT/cm2.err" FM_SPAWN_NO_GUARD=1 FM_HOME="$SM_HOME" FM_ROOT_OVERRIDE="$ROOT" \ - "$ROOT/bin/fm-spawn.sh" cm2 "$PROJ2" "sh -c 'echo sm-crew-ok'" --backend herdr \ + "$ROOT/bin/fm-spawn.sh" cm2 "$PROJ2" "sh -c 'echo sm-crew-ok'" --mode no-mistakes --yolo off --backend herdr \ >"$CM2_OUT" 2>"$CM2_ERR" rc=$? [ "$rc" -eq 0 ] || fail "a crewmate spawned FROM the secondmate-shaped home failed"$'\n'"--- stdout ---"$'\n'"$(cat "$CM2_OUT")"$'\n'"--- stderr ---"$'\n'"$(cat "$CM2_ERR")" diff --git a/tests/fm-backend-orca.test.sh b/tests/fm-backend-orca.test.sh index a54e448d108..5ee3fd570d5 100755 --- a/tests/fm-backend-orca.test.sh +++ b/tests/fm-backend-orca.test.sh @@ -454,7 +454,7 @@ test_spawn_preserves_orca_metadata_when_pathless_worktree_cleanup_fails() { out=$( PATH="$FB:$PATH" FM_ORCA_LOG="$LOG" FM_ORCA_RESPONSES="$RESP" \ FM_ROOT_OVERRIDE="$ROOT" FM_STATE_OVERRIDE="$state" FM_DATA_OVERRIDE="$data" FM_CONFIG_OVERRIDE="$config" \ FM_PROJECTS_OVERRIDE="$TMP_ROOT/unused-projects" FM_SPAWN_NO_GUARD=1 \ - "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --backend orca 2>&1 ) + "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --mode no-mistakes --yolo off --backend orca 2>&1 ) status=$? [ "$status" -ne 0 ] || fail "Orca spawn should fail when path parsing and cleanup fail" assert_contains "$out" "orca worktree create did not return a path" \ @@ -489,7 +489,7 @@ test_spawn_writes_orca_metadata_and_launches_harness() { out=$( PATH="$FB:$PATH" FM_ORCA_LOG="$LOG" FM_ORCA_RESPONSES="$RESP" \ FM_ROOT_OVERRIDE="$ROOT" FM_STATE_OVERRIDE="$state" FM_DATA_OVERRIDE="$data" FM_CONFIG_OVERRIDE="$config" \ FM_PROJECTS_OVERRIDE="$TMP_ROOT/unused-projects" FM_SPAWN_NO_GUARD=1 \ - "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --backend orca 2>&1 ) + "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --mode no-mistakes --yolo off --backend orca 2>&1 ) expect_code 0 $? "fm-spawn.sh --backend orca should succeed with fake Orca"$'\n'"$out" assert_contains "$out" "spawned $id harness=claude kind=ship mode=no-mistakes yolo=off window=fm-$id worktree=$wt" \ "spawn output missing Orca window/worktree summary" @@ -551,7 +551,7 @@ test_spawn_refuses_orca_when_runtime_not_ready() { out=$( PATH="$FB:$PATH" FM_ORCA_LOG="$LOG" FM_ORCA_RESPONSES="$RESP" FM_ORCA_STATUS_RESPONSE=sequence \ FM_ROOT_OVERRIDE="$ROOT" FM_STATE_OVERRIDE="$state" FM_DATA_OVERRIDE="$data" FM_CONFIG_OVERRIDE="$config" \ FM_PROJECTS_OVERRIDE="$TMP_ROOT/unused-projects" FM_SPAWN_NO_GUARD=1 \ - "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --backend orca 2>&1 ) + "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --mode no-mistakes --yolo off --backend orca 2>&1 ) status=$? [ "$status" -ne 0 ] || fail "fm-spawn.sh --backend orca should refuse when Orca runtime is not ready" assert_contains "$out" "requires a ready Orca runtime" \ @@ -582,7 +582,7 @@ test_spawn_refuses_orca_nonisolated_worktree() { out=$( PATH="$FB:$PATH" FM_ORCA_LOG="$LOG" FM_ORCA_RESPONSES="$RESP" \ FM_ROOT_OVERRIDE="$ROOT" FM_STATE_OVERRIDE="$state" FM_DATA_OVERRIDE="$data" FM_CONFIG_OVERRIDE="$config" \ FM_PROJECTS_OVERRIDE="$TMP_ROOT/unused-projects" FM_SPAWN_NO_GUARD=1 \ - "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --backend orca 2>&1 ) + "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --mode no-mistakes --yolo off --backend orca 2>&1 ) status=$? expect_code 1 "$status" "fm-spawn.sh --backend orca should refuse a primary checkout worktree" assert_contains "$out" "orca worktree create did not yield an isolated worktree" \ @@ -617,7 +617,7 @@ test_spawn_removes_orca_worktree_when_terminal_create_fails() { out=$( PATH="$FB:$PATH" FM_ORCA_LOG="$LOG" FM_ORCA_RESPONSES="$RESP" \ FM_ROOT_OVERRIDE="$ROOT" FM_STATE_OVERRIDE="$state" FM_DATA_OVERRIDE="$data" FM_CONFIG_OVERRIDE="$config" \ FM_PROJECTS_OVERRIDE="$TMP_ROOT/unused-projects" FM_SPAWN_NO_GUARD=1 \ - "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --backend orca 2>&1 ) + "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --mode no-mistakes --yolo off --backend orca 2>&1 ) status=$? [ "$status" -ne 0 ] || fail "Orca spawn should fail when terminal creation fails" assert_absent "$state/$id.meta" "terminal-create abort should not record metadata after successful cleanup" @@ -651,7 +651,7 @@ test_spawn_preserves_orca_metadata_when_abort_cleanup_fails() { out=$( PATH="$FB:$PATH" FM_ORCA_LOG="$LOG" FM_ORCA_RESPONSES="$RESP" \ FM_ROOT_OVERRIDE="$ROOT" FM_STATE_OVERRIDE="$state" FM_DATA_OVERRIDE="$data" FM_CONFIG_OVERRIDE="$config" \ FM_PROJECTS_OVERRIDE="$TMP_ROOT/unused-projects" FM_SPAWN_NO_GUARD=1 \ - "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --backend orca 2>&1 ) + "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --mode no-mistakes --yolo off --backend orca 2>&1 ) status=$? [ "$status" -ne 0 ] || fail "Orca spawn should fail when terminal creation and abort cleanup fail" assert_contains "$(cat "$LOG")" $'orca\x1f''worktree'$'\x1f''rm'$'\x1f''--worktree'$'\x1f''id:wt-cleanup-fail'$'\x1f''--force'$'\x1f''--json' \ @@ -683,7 +683,7 @@ test_spawn_releases_orca_resources_when_metadata_write_fails() { out=$( PATH="$FB:$PATH" FM_ORCA_LOG="$LOG" FM_ORCA_RESPONSES="$RESP" \ FM_ROOT_OVERRIDE="$ROOT" FM_STATE_OVERRIDE="$state" FM_DATA_OVERRIDE="$data" FM_CONFIG_OVERRIDE="$config" \ FM_PROJECTS_OVERRIDE="$TMP_ROOT/unused-projects" FM_SPAWN_NO_GUARD=1 \ - "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --backend orca 2>&1 ) + "$ROOT/bin/fm-spawn.sh" "$id" "$proj" claude --mode no-mistakes --yolo off --backend orca 2>&1 ) status=$? [ "$status" -ne 0 ] || fail "Orca spawn should fail when metadata cannot be written" assert_contains "$out" "Is a directory" "spawn should fail at metadata publication" diff --git a/tests/fm-backend.test.sh b/tests/fm-backend.test.sh index 3052ebc4227..0abafbd6c0a 100755 --- a/tests/fm-backend.test.sh +++ b/tests/fm-backend.test.sh @@ -908,7 +908,7 @@ run_spawn_symlink_case() { #