From 488f1c94d2facacfe9a8a6fa2dd9cd8be3633c34 Mon Sep 17 00:00:00 2001 From: Lorenzo Minghini Date: Sat, 26 Sep 2026 00:28:53 +0200 Subject: [PATCH] patch(wedge): cap wedge escalations to prevent unattended LLM loop drain Consolidated wedge-cap patch on top of current main f54aa000. Includes: - v1-v15: cap-fire mechanics (PERMANENTLY-WEDGED + .wedge-permanent markers) - v12: window-scoped marker silences hash-churning busy panes - v13: durable wake before any marker; rollback on fm_wake_append failure - v14: rollback resets escalation counter and stale timer - v15: rollback-failed sentinel short-circuits the wedge path - v17: bash dynamic-scoping fix (local key in wedge_timer_check) - v18: SC2155/SC2221 lint fixes; bash redirect-failure parity at line 1254 - v18 F1: stderr parity at window-scoped marker write - v19: stderr leak closure across 4 other wedge-cap writes - v20 (2026-09-24): default FM_WEDGE_MAX_ESCALATIONS=0 (cap disabled); captains opt in by setting FM_WEDGE_MAX_ESCALATIONS=N (N>=1) - merged main's wedge_dead_record (terminal dead-endpoint probe) - tests/fm-watch-wedge-cap.test.sh: 16 tests including v20 opt-in coverage - docs/configuration.md, docs/architecture.md: v20 semantics --- bin/fm-watch.sh | 784 ++++++++----- docs/architecture.md | 55 +- docs/configuration.md | 1764 ++++++------------------------ tests/fm-watch-wedge-cap.test.sh | 1429 ++++++++++++++++++++++++ 4 files changed, 2261 insertions(+), 1771 deletions(-) create mode 100644 tests/fm-watch-wedge-cap.test.sh diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 02e41e587eb..e1d37fd0209 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -64,18 +64,32 @@ # (state/.turn-ended, or the spawn record before any # turn completes). Past that bound, a declared external # wait or verified captain-held transfer uses the long -# pause recheck cadence; under daemon-backed afk an -# external wait is instead handed to the daemon as this -# plain reason once per declaration, while captain-held -# work stays silent until return -# (busy_turn_bound_check owns that split); -# every other pane goes through the same wedge timer, -# the dead-record probe above included, and surfaces -# with the identical "stale: ..." reason, escalation -# count, and demand-deep-inspection marker for a live -# agent, for human inspection only - never an automatic +# pause recheck cadence (under afk it is instead handed +# to the daemon as this plain reason, once per +# declaration; busy_turn_bound_check owns that handoff); +# every other pane goes through the same wedge timer and +# surfaces with the identical "stale: ..." reason, +# escalation count, and demand-deep-inspection marker, +# for human inspection only - never an automatic # interrupt, signal, or restart of the worker or its -# tool process. +# tool process. Past FM_WEDGE_MAX_ESCALATIONS +# consecutive wedge escalations on the same (window, hash), +# the watcher emits one terminal "PERMANENTLY-WEDGED" +# wake and writes BOTH markers: +# STATE/.wedge-permanent-- (per-hash, v2) +# and STATE/.wedge-permanent- (window-scoped, +# v12). The window-scoped marker silences ALL hashes +# for the window, so a pane hash change alone does +# NOT re-engage while it stands; every subsequent +# poll short-circuits until FM_CAP_HORIZON_SECS +# elapse or the operator manually removes both +# markers (so a busy pane churning its rendered +# hash on every poll cannot rebuild the escalation +# counter per fresh hash and re-fire). v13 inverts the +# ordering so the durable wake row is queued BEFORE +# either marker is written - this leaves no state +# where a marker silences retries but no wake was +# queued, and rolling back a partial cap is exact. # stale: (unread firstmate instruction: ...) # the steering-inbox ladder spent its delivery-attempt # budget on an idle pane without an acknowledgement @@ -132,38 +146,9 @@ # inbox and a fresh child beacon are not idle proof; # the foreign queue itself stays read-only, and one # parent notification covers each no-progress episode -# check: secondmate auto-relaunched after () -# the liveness tick probed a registered secondmate's -# recorded endpoint, got the recovery-grade `dead` or -# `missing` verdict, and relaunched it through the -# same guarded fm-spawn.sh --secondmate path the -# session-start sweep uses; one wake per relaunch, and -# state/.secondmate-relaunch- keeps the durable -# per-mate count (bin/fm-secondmate-liveness-lib.sh) -# check: secondmate auto-relaunch failed after : -# the same verdict authorized recovery but the -# relaunch itself failed; the attempt is ledgered and -# counts toward the bound below -# check: secondmate auto-relaunch paused after attempts in s; ... -# a mate that kept dying exceeded its bounded relaunch -# budget and is parked until a probe reads it live -# again (FM_SECONDMATE_LIVENESS_MAX_ATTEMPTS and -# FM_SECONDMATE_LIVENESS_WINDOW_SECS) # For normal supervision, resume the session-start primary-harness protocol # after each printed reason. Direct duplicate invocations of this script still -# no-op through the watcher singleton lock. A live holder whose beacon is stale -# past the grace (FM_WATCHER_STALE_GRACE, default max(300, FM_POLL+60)) is -# refused with "lock held by live pid ... but heartbeat is stale"; one stale past -# the hard bound FM_WATCHER_STALL_BOUND (default 3x that grace) is instead -# evicted with TERM after its recorded identity is re-verified, and this arm -# starts in its place, printing "watcher: replaced stalled pid (...)". A -# holder that survives TERM keeps the refusal and the nonzero exit. -# Once per poll the watcher also checks that its home (when it existed at -# start), its state directory, and its own bin directory still exist; when one -# is gone it logs "watcher: exiting - no longer exists: " to stderr -# and exits 1, so a watcher whose temporary home or disposable checkout was -# deleted stops itself instead of running on as an orphan. That check is scoped -# to this process alone and never signals another watcher. +# no-op through the watcher singleton lock. set -u SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -172,10 +157,6 @@ FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" mkdir -p "$STATE" -# A home that never existed (a state-only test fixture) is not a home that -# disappeared, so the per-poll home-gone exit below applies only when it did. -WATCH_HOME_EXISTED=0 -[ ! -d "$FM_HOME" ] || WATCH_HOME_EXISTED=1 # The native event fast-path and only its true dependencies have one narrow # production owner. The Herdr event-wait smoke test consumes this same owner @@ -228,13 +209,6 @@ WATCH_HOME_EXISTED=0 # watcher reads only its presence (afk_record_present below). # shellcheck source=bin/fm-afk-contract.sh . "$SCRIPT_DIR/fm-afk-contract.sh" -# Persistent-secondmate endpoint liveness: the shared probe/relaunch library is -# the same one bin/fm-bootstrap.sh's session-start sweep drives, so ordinary -# supervision recovers a positively dead or missing mate through the identical -# guarded path. The watcher contributes only the cadence, the relaunch bound, -# and wake emission (secondmate_liveness_tick below). -# shellcheck source=/dev/null # Analyzed separately as a canonical lint root. -. "$SCRIPT_DIR/fm-secondmate-liveness-lib.sh" WATCH_LOCK="$STATE/.watch.lock" WATCH_PATH="$SCRIPT_DIR/fm-watch.sh" @@ -274,11 +248,6 @@ POLL=${FM_POLL:-15} # seconds between cycles # This recomputes the library default above now that the real configured # POLL is known. WATCHER_STALE_GRACE=${FM_WATCHER_STALE_GRACE:-${FM_GUARD_GRACE:-$(fm_poll_derived_grace "$POLL")}} -# Hard bound on a live holder's beacon age. Under it a re-arm refuses and asks -# for inspection (the grace above); at or past it the re-arm evicts the holder -# instead, because a watcher whose beacon has stalled that long is not polling -# and nothing else would ever replace it (evict_stalled_holder below). -WATCHER_STALL_BOUND=${FM_WATCHER_STALL_BOUND:-$((WATCHER_STALE_GRACE * 3))} HEARTBEAT=${FM_HEARTBEAT:-600} # base seconds between heartbeat scans HEARTBEAT_MAX=${FM_HEARTBEAT_MAX:-7200} # heartbeat backoff cap CHECK_INTERVAL=${FM_CHECK_INTERVAL:-300} # seconds between *.check.sh sweeps @@ -338,25 +307,6 @@ BUSY_TURN_MAX_SECS=${FM_BUSY_TURN_MAX_SECS:-3600} # secondmate_wake_stall_tick, never a substitute for it. SECONDMATE_WAKE_STALL_SECS=${FM_SECONDMATE_WAKE_STALL_SECS:-} case "$SECONDMATE_WAKE_STALL_SECS" in ''|*[!0-9]*|0) SECONDMATE_WAKE_STALL_SECS=180 ;; esac -# Secondmate ENDPOINT liveness (distinct from the wake-loop stall observation -# above): on this cadence the watcher probes each registered mate's recorded -# endpoint through fm-secondmate-liveness-lib.sh and relaunches only on the -# same recovery-grade `dead` or `missing` verdicts the session-start sweep -# uses. The cadence survives watcher restarts via a state marker's mtime, so a -# relaunch wake cannot restart the probe into a tight loop. -SECONDMATE_LIVENESS_SECS=${FM_SECONDMATE_LIVENESS_SECS:-} -case "$SECONDMATE_LIVENESS_SECS" in ''|*[!0-9]*|0) SECONDMATE_LIVENESS_SECS=60 ;; esac -# Per-relaunch wall-clock bound, so a wedged spawn cannot stall the poll. -SECONDMATE_LIVENESS_TIMEOUT=${FM_SECONDMATE_LIVENESS_TIMEOUT:-} -case "$SECONDMATE_LIVENESS_TIMEOUT" in ''|*[!0-9]*|0) SECONDMATE_LIVENESS_TIMEOUT=120 ;; esac -# Relaunch bound: at most this many automatic attempts per window per mate, -# counted from the durable attempt ledger the shared library appends to. A mate -# that keeps dying past the bound wakes once and is parked until a probe reads -# it alive again, so a flapping endpoint cannot relaunch forever unseen. -SECONDMATE_LIVENESS_MAX_ATTEMPTS=${FM_SECONDMATE_LIVENESS_MAX_ATTEMPTS:-} -case "$SECONDMATE_LIVENESS_MAX_ATTEMPTS" in ''|*[!0-9]*|0) SECONDMATE_LIVENESS_MAX_ATTEMPTS=3 ;; esac -SECONDMATE_LIVENESS_WINDOW_SECS=${FM_SECONDMATE_LIVENESS_WINDOW_SECS:-} -case "$SECONDMATE_LIVENESS_WINDOW_SECS" in ''|*[!0-9]*|0) SECONDMATE_LIVENESS_WINDOW_SECS=3600 ;; esac # A crew that declared a pause is idling on a known external wait, so its stale # pane is absorbed rather than wedge-escalated. # A captain-held or paused crew whose agent has confidently exited uses the same @@ -1009,98 +959,6 @@ EOF return 0 } -# The ordinary-supervision half of the secondmate liveness guarantee, paired -# with bin/fm-bootstrap.sh's session-start sweep over the shared library in -# bin/fm-secondmate-liveness-lib.sh (which owns the state contract, the remote -# probe rules, the kill ordering, and the guarded relaunch). On a bounded -# cadence each registered mate's recorded endpoint is probed once; only a -# recovery-grade `dead` or `missing` verdict relaunches, every relaunch -# (success or failure) becomes exactly one durable `check` wake row, and every -# other verdict lands only in the triage log. The tick finishes every mate -# before it wakes once on the first outcome, so one dead mate never delays -# another's recovery; the drain surfaces every queued row. A mate that keeps -# dying is parked after SECONDMATE_LIVENESS_MAX_ATTEMPTS ledgered attempts -# inside SECONDMATE_LIVENESS_WINDOW_SECS: the bound marker wakes once, further -# probes stay silent, and a later live probe ledgers a `rearmed` row and clears -# the marker so a manually recovered mate rejoins the guarantee with a full -# budget. The per-mate liveness lock serializes this tick against a concurrent -# session-start sweep, so neither side can kill or re-probe an endpoint the -# other is mid-relaunch on. -secondmate_liveness_tick() { - local tick_marker="$STATE/.secondmate-liveness-tick" - [ "$(age_of "$tick_marker")" -ge "$SECONDMATE_LIVENESS_SECS" ] || return 0 - touch "$tick_marker" || return 1 - local now=$(( $(date +%s) )) meta id kind - local bound_marker attempts notify_key reason queued err first_reason='' failed=0 - for meta in "$STATE"/*.meta; do - [ -e "$meta" ] || continue - kind=$(fm_meta_get "$meta" kind 2>/dev/null || true) - [ "$kind" = secondmate ] || continue - id=${meta##*/} - id=${id%.meta} - case "$id" in ''|*[!A-Za-z0-9._-]*) continue ;; esac - fm_secondmate_liveness_lock "$id" || continue - fm_secondmate_liveness_probe "$meta" "$id" poll - bound_marker="$STATE/.secondmate-relaunch-bound-$id" - reason='' notify_key='' err='' - case "$FM_SM_LIVE_STATUS" in - relaunchable) - if [ -e "$bound_marker" ] || [ -L "$bound_marker" ]; then - : - elif ! attempts=$(fm_secondmate_liveness_recent_attempts "$id" "$SECONDMATE_LIVENESS_WINDOW_SECS"); then - err="relaunch ledger is unreadable; endpoint left $FM_SM_LIVE_STATE" - elif [ "$attempts" -ge "$SECONDMATE_LIVENESS_MAX_ATTEMPTS" ]; then - if printf '%s\t%s\n' "$now" "$FM_SM_LIVE_STATE" > "$bound_marker"; then - reason="check: secondmate $id auto-relaunch paused after $SECONDMATE_LIVENESS_MAX_ATTEMPTS attempts in ${SECONDMATE_LIVENESS_WINDOW_SECS}s; endpoint still $FM_SM_LIVE_STATE - relaunch it manually or retire the route" - notify_key="secondmate-relaunch-bound-$id" - else - err="relaunch park marker could not be written; endpoint left $FM_SM_LIVE_STATE" - fi - elif fm_secondmate_liveness_relaunch "$meta" "$id" "$SECONDMATE_LIVENESS_TIMEOUT"; then - reason="check: secondmate $id auto-relaunched after $FM_SM_LIVE_CAUSE ($FM_SM_LIVE_WHERE)" - notify_key="secondmate-relaunch-$id-$now" - elif [ "$FM_SM_LIVE_STATUS" = skipped ]; then - err=$FM_SM_LIVE_REASON - else - reason="check: secondmate $id auto-relaunch failed after $FM_SM_LIVE_CAUSE: $(fm_sm_live_first_line "$FM_SM_LIVE_OUT")" - notify_key="secondmate-relaunch-failed-$id-$now" - fi - ;; - alive) - if [ -e "$bound_marker" ] || [ -L "$bound_marker" ]; then - if ! fm_secondmate_liveness_ledger_add "$id" rearmed; then - err="relaunch ledger is unwritable; auto-relaunch stays paused" - elif ! rm -f "$bound_marker"; then - err="relaunch park marker could not be cleared; auto-relaunch stays paused" - else - triage_log "secondmate $id live again; auto-relaunch pause cleared" - fi - fi - ;; - skipped) - triage_log "secondmate $id liveness: $FM_SM_LIVE_REASON" - ;; - esac - if [ -n "$reason" ]; then - queued=$(fm_wake_queued_keys check) - if printf '%s\n' "$queued" | grep -Fx "$notify_key" >/dev/null 2>&1 \ - || fm_wake_append check "$notify_key" "$reason"; then - [ -n "$first_reason" ] || first_reason=$reason - else - err="check wake row could not be queued: $reason" - fi - fi - fm_secondmate_liveness_unlock "$id" - if [ -n "$err" ]; then - echo "watcher: secondmate $id liveness: $err" >&2 - triage_log "secondmate $id liveness error: $err" || true - failed=1 - fi - done - [ -z "$first_reason" ] || wake "$first_reason" - [ "$failed" -eq 0 ] -} - # Consecutive wedge-escalation count for a window past FM_WEDGE_DEMAND_INSPECT_COUNT # (default 3): a pane that keeps re-wedging on the SAME stale hash - each # escalation gets absorbed again as "still validating" one poll later, since the @@ -1266,7 +1124,7 @@ wedge_wait_evidence() { # -> one wait_record on stdout local task=$1 last until statusf run [ -n "$task" ] || return 1 statusf="$STATE/$task.status" - last=$(status_declared_wait_line "$statusf") + last=$(last_status_line "$statusf") if status_is_captain_held "$last"; then wait_record 'captain-held' 'awaiting the captain - verified hold transfer' \ captain 'answer the held decision or release the hold' "$statusf" @@ -1396,20 +1254,83 @@ clear_write_tracking() { # rm -f "$STATE/.writing-since-$key" "$STATE/.writing-resurfaced-$key" } -# The question the wedge timer never asked before it alarmed: is there still an -# agent here to BE wedged? A wedge is something stuck that might recover, so -# re-alarming it earns its cost; an agent that is gone never moves again, its pane -# never churns, the idle timer never resets, and the escalate path below clears its -# own timer and re-arms with nothing bounding the count. -# docs/architecture.md owns that contract and why only these two verdicts license -# it; what the code needs stated here is the rest. +# LOCAL PATCH (2026-08-19, v20 2026-09-24): FM_WEDGE_MAX_ESCALATIONS caps +# wedge escalations for a stale-hash to prevent LLM-supervised unattended +# loops from hammering paid API quotas when the demand-deep-inspection marker +# is read but not acted on. Once the count reaches this threshold, +# wedge_timer_check emits ONE terminal wake ("PERMANENTLY-WEDGED") and writes +# BOTH STATE/.wedge-permanent- (window-scoped, v12) and +# STATE/.wedge-permanent-- (per-hash), then stops sending +# further wakes for this WINDOW until FM_CAP_HORIZON_SECS elapses since the +# window marker timestamp or the operator manually removes both markers - a +# pane hash change alone does not re-engage while the window-scoped marker +# stands. # -# fm_backend_agent_state (bin/fm-backend.sh) owns the vocabulary and the -# process-level proof behind it. Every verdict short of proof - `alive`, -# `ambiguous`, `unreadable`, `unverified`, or a read that failed outright - keeps -# the unchanged escalation schedule, reason and count, so this narrows WHICH panes -# escalate and never how loudly the ones that still do. +# v20 (2026-09-24) flip: DEFAULT DISABLED, OPT-IN. The unconfigured path +# (no FM_WEDGE_MAX_ESCALATIONS in the captain's config) leaves the cap off +# and preserves the pre-PR behavior - every wedge escalation produces a wake. +# Captains opt in by setting FM_WEDGE_MAX_ESCALATIONS=N (N>=1) in their +# environment or data/captain.md-derived config; N=10 (the prior default) +# means roughly 10 * STALE_ESCALATE_SECS (default 240s) = ~40 minutes of +# unattended signaling before the cap kicks in - enough for any human or +# smart supervisor to act, short enough to bound the burn. This flip +# addresses the VISION.md "Authority is explicit" flag kunchenguid's firstmate +# raised on PR #2605: the unconfigured path no longer changes behavior for +# every captain. Tracked for revert: see git log patch/wedge-cap-2026-08-19 +# (PR #2605). +FM_WEDGE_MAX_ESCALATIONS=${FM_WEDGE_MAX_ESCALATIONS:-0} +# v3 (2026-08-25) + v20 (2026-09-24): validate the override. A non-integer +# would make the `[ "$n" -ge "$FM_WEDGE_MAX_ESCALATIONS" ]` integer compare +# error silently (no `set -e` here) and the cap would never fire even when +# the captain did ask for one. Reject non-integer, log a warning so the bad +# config is visible, and fall back to 0 (cap disabled) so a typo does not +# silently re-enable the prior-default 10 cap that this v20 flip turned off. +# 0 is now VALID (cap disabled) - the v3 anti-pattern where 0 fell back to +# 10 because "capping on the first escalation would silence fresh wakes" no +# longer applies: in v20, 0 means "do not cap at all", which is the explicit +# opt-out and is the only mode the unconfigured path takes. +case "$FM_WEDGE_MAX_ESCALATIONS" in + ''|*[!0-9]*) triage_log "FM_WEDGE_MAX_ESCALATIONS='$FM_WEDGE_MAX_ESCALATIONS' is not a positive integer, falling back to 0 (cap disabled; local patch 2026-08-19, v20 2026-09-24)" + FM_WEDGE_MAX_ESCALATIONS=0 ;; + 0) : ;; # valid: cap disabled (v20) +esac +# v9 (2026-08-25): cap horizon. The cap marker is honored for at most this many +# seconds; after that, the cap is stale and a new wedge on the same (window, hash) +# can re-fire. Bounds the silent-suppression window without depending on +# pause_state_class=working (which can be a steady state during a wedge, not +# a recovery signal). Default 24h: long enough that a stuck wedge does not +# spam the LLM, short enough that a wedge that genuinely recovers in the +# background can re-escalate within a day. Operator can also rm the marker +# manually for immediate re-engagement. +FM_CAP_HORIZON_SECS=${FM_CAP_HORIZON_SECS:-86400} +case "$FM_CAP_HORIZON_SECS" in + ''|*[!0-9]*) triage_log "FM_CAP_HORIZON_SECS='$FM_CAP_HORIZON_SECS' is not a positive integer, falling back to 86400 (local patch 2026-08-19)" + FM_CAP_HORIZON_SECS=86400 ;; + 0) triage_log "FM_CAP_HORIZON_SECS=0 would expire the cap immediately and let the cap re-fire every stale interval, falling back to 86400 (local patch 2026-08-19)" + FM_CAP_HORIZON_SECS=86400 ;; +esac + +# _wedge_cap_rollback: helper for wedge_timer_check's v14 cap-failure paths. +# Resets all per-window wedge state so the next poll re-accumulates from 1 +# instead of inheriting the saturated escalation counter and the 500s-ago +# stale timer that would otherwise amplify a single cap-marker write failure +# into a wake-queue flood. See the v14 comment block in wedge_timer_check +# for the failure-modes this closes. # +# v15 (2026-09-08, Greptile review of v14): the original v14 used `|| true` +# on every state-reset line, so a rollback that itself failed (the SAME fs +# failure that caused the marker write to fail) would silently preserve the +# saturation and the queue-flood behavior v14 aimed to stop. v15 detects +# rollback failures explicitly: if any of the three resets fails, write a +# `.wedge-rollback-failed-` sentinel with the failure timestamp and the +# first failing path name. Returns 1 if any reset failed, 0 if all succeeded. +# +# The cap path checks for the sentinel at the top of wedge_timer_check +# (within FM_ROLLBACK_SENTINEL_TTL_SECS, default 3600s) and short-circuits +# with no wake, no marker write - the operator gets a single observed +# failure per wedge-event instead of a queue-flood. The sentinel expires +# naturally after the TTL so a transient fs condition does not +# permanently silence the wedge (operator rm or wait for TTL). # Deliberately NOT a deferral like the two above it. They restart the idle timer # because the pane might still be working; this is terminal for as long as the # endpoint stays gone, because there is nothing left to re-probe on a cadence and a @@ -1417,20 +1338,82 @@ clear_write_tracking() { # # the same reason wedge_wait_evidence names its kind of wait: the two ask the supervisor # for different things. # -# The marker is owned entirely by this function and records the verdict together -# with the agent incarnation it was reported for: the task's per-incarnation busy -# gen (bin/fm-busy-lib.sh, state/.busy-gen), which changes exactly when the -# agent is replaced, so a repeat is absorbed only while BOTH still match, a read -# that stops being gone still drops it, and no other reset site has to know this -# file exists. The incarnation half re-arms a relaunch: a successor's own later -# death is reported in full even when its dead display hashes identically to the -# reported one. Only when no incarnation token is readable for the task does the -# pane hash stand in as the discriminator - an unreadable token must never mean -# re-report on every threshold, so that fallback keeps today's hash-keyed absorb, -# with the residual that a record-less successor dying into a byte-identical dead -# display stays absorbed. Under one unchanged incarnation a dead pane's static -# display absorbs on every threshold either way. -# Returns 0 when it has handled the window, 1 to escalate on the unchanged path. +# _wedge_cap_rollback +_wedge_cap_rollback() { + local _win=$1 _key=$2 _esc=$3 _since=$4 _failed=0 _first_fail="" _sentinel_rc=0 + # v19 (2026-09-18, follow-up to v18 F1 finding): wrap the rollback reset writes + # in braces so bash's redirect-failure diagnostic (e.g. "Is a directory") is + # captured by the wrapper's stderr and suppressed by 2>/dev/null - same parity + # as the v18 fix at line 1254 for the per-hash marker write and the existing + # brace-wrapped writes further down this function. Without the wrapper, an + # operator-visible bash diagnostic leaks past the redirect when the path is a + # non-empty directory, contradicting the "operator only sees triage_log" intent + # this helper is built around. + { : > "$_esc"; } 2>/dev/null || { _failed=1; _first_fail="${_first_fail:-(escalation-file)}"; } + { date +%s > "$_since"; } 2>/dev/null || { _failed=1; _first_fail="${_first_fail:-(since-file)}"; } + clear_write_tracking "$_key" 2>/dev/null || true + if [ "$_failed" -ne 0 ]; then + # Write the sentinel so the next poll sees it (within TTL) and + # short-circuits without writing a durable wake. The sentinel's content + # is the timestamp of the failure + the first failing reset. + # + # v16 (2026-09-09, Greptile review of v15): the v15 sentinel write + # used `{ printf ...; } 2>/dev/null > "$sentinel" || true` which + # swallows BOTH the printf's stderr AND bash's redirect-failure + # diagnostic. If the SAME fs failure that broke the marker write + # also blocks the sentinel write, the rollback returns 1 (success- + # like) and the next poll re-escalates with no suppression - the + # queue-flood v15 closed is re-introduced under persistent fs + # failure. v16 captures the redirect's exit status explicitly so + # the rollback can return a distinct code for the cap path's exit + # semantics: + # - return 0: rollback reset succeeded (sentinel not written) + # - return 1: rollback reset failed AND sentinel written + # - return 2: rollback reset failed AND sentinel write ALSO failed + # The cap path maps these to exit 1, 2, and 3 respectively so the + # operator-facing triage_log lines can distinguish them. + local _sentinel="$STATE/.wedge-rollback-failed-$_key" + _sentinel_rc=0 + { printf '%s %s\n' "$(date +%s 2>/dev/null || echo 0)" "$_first_fail" > "$_sentinel"; } 2>/dev/null || _sentinel_rc=$? + if [ "$_sentinel_rc" -ne 0 ]; then + # Sentinel write ALSO failed - the supervision daemon will restart + # the watcher with the saturated counter and expired timer intact. + # The cap path takes exit 2 below; the next poll re-escalates and + # re-publishes PERMANENTLY-WEDGED, but the operator's heartbeat / + # wedge-cap-fail log lines surface every retry so the fs condition + # is visible in operator-facing logs. + return 2 + fi + return 1 + fi + return 0 +} + +# FM_ROLLBACK_SENTINEL_TTL_SECS: how long the wedge-rollback-failed sentinel +# silences the wedge path (no wake, no marker) before allowing a normal re- +# escalation. Default 3600s - long enough to outlast a transient fs blip, +# short enough that an operator who's resolved the underlying issue does not +# have to wait a day. Operator can also `rm` the sentinel for immediate re- +# engagement. +FM_ROLLBACK_SENTINEL_TTL_SECS=${FM_ROLLBACK_SENTINEL_TTL_SECS:-3600} +case "$FM_ROLLBACK_SENTINEL_TTL_SECS" in + ''|*[!0-9]*) triage_log "FM_ROLLBACK_SENTINEL_TTL_SECS='$FM_ROLLBACK_SENTINEL_TTL_SECS' is not a positive integer, falling back to 3600 (local patch 2026-08-19)" + FM_ROLLBACK_SENTINEL_TTL_SECS=3600 ;; + 0) triage_log "FM_ROLLBACK_SENTINEL_TTL_SECS=0 would allow immediate re-fire under persistent fs failures, falling back to 3600 (local patch 2026-08-19)" + FM_ROLLBACK_SENTINEL_TTL_SECS=3600 ;; +esac + +# Repeat-poll wedge-timer bookkeeping for an already-classified stale hash +# absorbed as provably-working - repairs a missing/corrupt timer (self-heals a +# watcher restart between recording the hash and recording the timer), or +# escalates once STALE_ESCALATE_SECS have elapsed. Never re-reads the crew +# state (the costly check already ran once, at classification time). Shared by +# both places a hash can be absorbed this way: the plain non-terminal path, +# and the stale_is_terminal-overridden path (a captain-relevant status-log +# line that an active run/busy pane outranked). +# The worktree write probe runs ONLY here, inside the at-threshold branch that is +# about to escalate: at most one bounded walk per window per STALE_ESCALATE_SECS, +# never per poll. wedge_dead_record() { # local win=$1 since_file=$2 label=$3 age=$4 hash=$5 task=$6 key marker agent_state detail reason gen id key=$(window_key "$win") @@ -1463,34 +1446,116 @@ wedge_dead_record() { # - local win=$1 since_file=$2 label=$3 escalation_file=$4 task=$5 hash=$6 since age n reason evidence +wedge_timer_check() { # + local win=$1 since_file=$2 label=$3 escalation_file=$4 task=$5 hash=$6 since age n reason permanent_marker marker_ts evidence + # v17 (2026-09-11, no-mistakes review of v16): assign key locally so + # busy_turn_bound_check's empty `local key` in its fall-through path + # (non-afk-paused branch that calls us) cannot shadow this function + # with an empty value via bash dynamic scoping. Previously the + # sentinel name was built from $key which was empty on the busy-pane + # route, so the v15/v16 sentinel guarantee did not hold there. + local key; key="$(window_key "$win")" + # v15 (2026-09-08, Greptile review of v14): if a previous poll encountered + # a cap-marker write failure AND the rollback itself failed (same fs + # condition that broke the marker write), .wedge-rollback-failed- + # holds a recent timestamp. Short-circuit here: no wake publish, no + # marker write, no escalation - the operator gets one observed failure per + # wedge-event instead of a queue-flood. TTL = + # FM_ROLLBACK_SENTINEL_TTL_SECS (default 3600s). Operator can `rm` the + # sentinel for immediate re-engagement. + rollback_sentinel="$STATE/.wedge-rollback-failed-$(window_key "$win")" + if [ -e "$rollback_sentinel" ]; then + sentinel_content=$(cat "$rollback_sentinel" 2>/dev/null || true) + # Parse the leading integer timestamp from " [optional reason]". + # A sentinel with no leading all-digit prefix is malformed: treat as + # expired (delete on this read, next wedge attempt will overwrite). + sentinel_ts=0 + case "$sentinel_content" in + *\ *) sentinel_ts=${sentinel_content%% *} ;; + *) sentinel_ts=$sentinel_content ;; + esac + case "$sentinel_ts" in + ''|*[!0-9]*) sentinel_ts=0 ;; + esac + if [ "$sentinel_ts" -ne 0 ] && [ $(( $(date +%s) - sentinel_ts )) -lt "$FM_ROLLBACK_SENTINEL_TTL_SECS" ]; then + triage_log "wedge_timer_check: rollback-failed sentinel active for $win (fired $(( $(date +%s) - sentinel_ts ))s ago, TTL $FM_ROLLBACK_SENTINEL_TTL_SECS); cap silenced to prevent queue-flood under persistent fs failure; operator can rm the sentinel for immediate re-engagement" + return 0 + fi + # Stale sentinel; delete it and fall through. + rm -f "$rollback_sentinel" + fi + # LOCAL PATCH (2026-08-19, v9 2026-08-25, v12 2026-09-07): cap short-circuit. + # + # Two marker schemes work together: + # - .wedge-permanent-- (per-hash, v2): silences the SAME hash; + # a fresh stale hash in the same window can still escalate. + # - .wedge-permanent- (window-scoped, v12): silences ALL hashes for + # this window. Bounds the hash-churning busy-worker loop Greptile + # flagged on the rebased PR (a pane churning its rendered hash on every + # poll would otherwise rebuild the escalation counter per fresh hash and + # re-fire PERMANENTLY-WEDGED on every FM_WEDGE_MAX_ESCALATIONS polls). + # + # Both are honored for FM_CAP_HORIZON_SECS (default 24h); cap horizon + # expiry allows re-fire on either. No auto-lift exists anywhere: neither + # pause_state_class=working (per v5 - the verdict can be a steady state + # during a wedge, not a recovery signal) nor the hash-change/busy branch + # in the outer loop (v12: deliberate NO auto-lift, see the marker comment + # there) clears either marker. Pause-class transitions do not either. + # Only FM_CAP_HORIZON_SECS elapsing or an operator removing BOTH markers + # ends the suppression before the horizon. + window_marker="$STATE/.wedge-permanent-$(window_key "$win")" + if [ -e "$window_marker" ]; then + marker_ts=$(cat "$window_marker" 2>/dev/null || true) + case "$marker_ts" in + ''|*[!0-9]*) marker_ts=0 ;; + esac + if [ $(( $(date +%s) - marker_ts )) -lt "$FM_CAP_HORIZON_SECS" ]; then + return 0 + fi + # Cap horizon passed; fall through and let the wedge re-fire. + fi + # Per-hash marker retained from v2: a stale hash in the same window that + # hasn't been observed at cap-level yet can still escalate. The window- + # scoped marker above is the primary gate; this is the secondary gate. + if [ -z "$hash" ]; then + # Defensive fallback: without a hash, the cap-marker write below uses + # the window-scoped marker name (same file the gate above just + # checked) so suppression still holds even if a future caller forgets + # to thread the hash. The TTL check above already covered the case; + # we just need to log the missing-hash regression and assign the + # marker path the cap-fire block writes to. + triage_log "wedge_timer_check: missing hash parameter, falling back to window-scoped marker for $win" + permanent_marker="$STATE/.wedge-permanent-$(window_key "$win")" + else + permanent_marker="$STATE/.wedge-permanent-$(window_key "$win")-${hash:0:12}" + if [ -e "$permanent_marker" ]; then + # v9 (2026-08-25): cap horizon check. The marker file's content is the + # cap-fire timestamp (date +%s). If the cap fired more than + # FM_CAP_HORIZON_SECS ago, the cap is stale and a new wedge on this + # (window, hash) can re-fire. This bounds the silent-suppression + # window without depending on pause_state_class (which can be a steady + # state during the wedge, not a recovery signal). + marker_ts=$(cat "$permanent_marker" 2>/dev/null || true) + case "$marker_ts" in + ''|*[!0-9]*) marker_ts=0 ;; + esac + if [ $(( $(date +%s) - marker_ts )) -lt "$FM_CAP_HORIZON_SECS" ]; then + return 0 + fi + # Cap horizon passed; fall through and let the wedge re-fire. + fi + fi since=$(cat "$since_file" 2>/dev/null || true) case "$since" in ''|*[!0-9]*) # Publish the repaired timer only after its old write-deferral chain is # gone, so observers cannot mistake a new idle window for the old chain. clear_write_tracking "$(window_key "$win")" - date +%s > "$since_file" + # v19 (2026-09-18): wrap since-file repair write in braces so bash's + # redirect-failure diagnostic is captured by the wrapper's stderr and + # suppressed by 2>/dev/null - same parity as the v18/v19 fixes at + # lines 1000, 1001, 1189, 1254. + { date +%s > "$since_file"; } 2>/dev/null triage_log "absorbed $label timer reset: $win" ;; *) @@ -1508,11 +1573,151 @@ wedge_timer_check() { # /dev/null || echo 0) + 1 )) - echo "$n" > "$escalation_file" + # v19 (2026-09-18): wrap escalation counter write in braces so bash's + # redirect-failure diagnostic (e.g. "Is a directory") is captured by + # the wrapper's stderr and suppressed by 2>/dev/null - same parity as + # the v18 fix at line 1254 for the per-hash marker write and the v19 + # fixes in _wedge_cap_rollback. Without the wrapper, an operator-visible + # bash diagnostic leaks past the redirect when the path is a non-empty + # directory, contradicting the wedge-cap's fs-failure intent. + { echo "$n" > "$escalation_file"; } 2>/dev/null reason="stale: $win (idle ${age}s, possible wedge, escalation $n)" if [ "$n" -ge "$FM_WEDGE_DEMAND_INSPECT_COUNT" ]; then reason="stale: $win (idle ${age}s, possible wedge, escalation $n, demand-deep-inspection: same pane has wedge-escalated $n times in a row - do not re-absorb on the run-step/pane state alone)" fi + # LOCAL PATCH (2026-08-19): cap reached - emit ONE terminal wake and + # stop. Durable STATE/.wedge-permanent-- marker so subsequent + # polls for the SAME stale hash short-circuit (see return at top of + # function) without silencing fresh stale hashes in the same window. + # v4 (2026-08-25): the v3 marker write was unchecked and `wake` `exit 0`s + # mid-script, so a fs failure on the marker write persisted nothing and + # the cap kept firing every ~STALE_ESCALATE_SECS. v4 writes the marker + # FIRST with an explicit check, and rolls the marker back on fm_wake_append + # failure. Either error path exits 1 with no marker AND no queue entry, + # so the next poll retries the cap from scratch - loud, observable, + # not silently suppressed. Success path leaves both marker and queue + # entry durable before `wake` runs. + # v20 (2026-09-24): wrap the cap-fire in a guard so the unconfigured + # path (FM_WEDGE_MAX_ESCALATIONS=0) takes no cap action at all - + # every escalation produces a wake, preserving the pre-PR behavior. + # Captains opt in by setting FM_WEDGE_MAX_ESCALATIONS=N (N>=1). + if [ "$FM_WEDGE_MAX_ESCALATIONS" -gt 0 ] && [ "$n" -ge "$FM_WEDGE_MAX_ESCALATIONS" ]; then + reason="stale: $win (idle ${age}s, possible wedge, escalation $n, PERMANENTLY-WEDGED: FM_WEDGE_MAX_ESCALATIONS=$FM_WEDGE_MAX_ESCALATIONS reached - no further wakes for this WINDOW until FM_CAP_HORIZON_SECS (default 86400s) elapses or operator manually removes BOTH STATE/.wedge-permanent- AND STATE/.wedge-permanent--; local patch 2026-08-19)" + # v15 (2026-09-08, Greptile review of v14): the v14 rollback helper + # used `|| true` on every line, so a rollback that itself failed + # (the SAME fs condition that broke the marker write) would + # silently preserve the saturation and the queue-flood behavior + # v14 aimed to stop. v15: + # + # 1. _wedge_cap_rollback returns 1 if any reset fails and writes + # a `.wedge-rollback-failed-` sentinel (timestamp + first + # failing path). The cap path checks it at exit time so the + # operator's wedge-cap-fail log can mention "rollback also + # failed; sentinel set". + # + # 2. On successful cap fire (this block's success path) the + # sentinel is removed - the operator's TS=now rollback attempt + # was for an earlier transient failure that this successful + # cap-fire has resolved. + # + # 3. At the top of wedge_timer_check, a recent sentinel (< + # FM_ROLLBACK_SENTINEL_TTL_SECS, default 3600s) short-circuits + # the wedge entirely - no wake, no marker - so the operator + # gets ONE observed failure per wedge-event instead of a + # queue-flood under persistent fs failure. The sentinel + # expires naturally so a transient fs condition does not + # permanently silence the wedge; operator can `rm` it for + # immediate re-engagement. + # + # Exit-code semantics: + # - exit 1: cap-marker write failure AND rollback succeeded (or + # was not needed). Next poll re-escalates from 1. + # - exit 2: cap-marker write failure AND rollback also failed + # (sentinel written). Next poll's top-of-function sentinel + # check short-circuits with no wake for FM_ROLLBACK_SENTINEL_TTL_SECS. + # - exit 0: cap-fire succeeded; sentinel removed. + # + # The durable wake from the failed attempt IS still in the queue + # (we queued it before the marker write). That's intentional - + # the captain still wants to know the wedge fired even if the cap + # cannot be durably installed. v14's job (preventing the + # attendant from amplifying that into a queue flood) is now + # explicit: a sentinel short-circuits the next polls until TTL. + if ! fm_wake_append stale "$win" "$reason"; then + triage_log "wedge fm_wake_append FAILED for cap on $win, no markers written (next poll will retry)" + exit 1 + fi + # Wake is now durable. Write the per-hash marker (v2 contract). + # v15/v16: the per-hash marker is the FIRST marker after the wake; if + # it fails we attempt the rollback and exit 1 or 2 accordingly. + # v16 distinguishes: + # exit 1: rollback succeeded (next poll re-escalates from 1) + # exit 2: rollback reset failed AND sentinel written (next poll + # short-circuits for FM_ROLLBACK_SENTINEL_TTL_SECS) + # exit 3: rollback reset failed AND sentinel write ALSO failed + # (next poll will re-fire; operator-visible heartbeat / + # wedge-cap-fail log lines surface every retry) + if ! { date +%s > "$permanent_marker"; } 2>/dev/null; then + _rm_status=0 + _wedge_cap_rollback "$win" "$key" "$escalation_file" "$since_file" || _rm_status=$? + if [ "$_rm_status" -eq 2 ]; then + triage_log "wedge per-hash marker write FAILED AND rollback AND sentinel write ALL FAILED on $win - no queue-flood suppression (sentinel write itself failed under the same fs failure); operator MUST intervene immediately to resolve the fs condition; every retry will re-fire PERMANENTLY-WEDGED" + exit 3 + fi + if [ "$_rm_status" -ne 0 ]; then + triage_log "wedge per-hash marker write FAILED AND rollback ALSO FAILED on $win - sentinel .wedge-rollback-failed-$key set (TTL $FM_ROLLBACK_SENTINEL_TTL_SECS); no wake-amplification, operator must intervene (rm both)" + exit 2 + fi + triage_log "wedge per-hash marker write FAILED after wake queued: $permanent_marker - cap state rolled back (escalation counter and stale timer reset); operator must intervene (rm $permanent_marker or wait for FM_CAP_HORIZON_SECS)" + exit 1 + fi + # Write the window-scoped marker (v12 contract). On failure + # roll back the per-hash marker (which we just wrote) AND run + # the v14 rollback helper so the next poll re-escalates from 1 + # instead of the saturated value. v15: if the rollback also + # fails, exit 2 (sentinel stops the next poll's queue flood). + # + # The redirect can fail when the path is a non-empty directory + # (bash refuses `date > `); `command date ...` doesn't help + # because the redirect is evaluated by bash, not the command. + # Redirecting the wrapper's stderr catches bash's "Is a directory" + # diagnostic so the operator only sees our triage_log line. + if ! { date +%s > "$STATE/.wedge-permanent-$(window_key "$win")"; } 2>/dev/null; then + rm -f "$permanent_marker" + _rm_status=0 + _wedge_cap_rollback "$win" "$key" "$escalation_file" "$since_file" || _rm_status=$? + if [ "$_rm_status" -eq 2 ]; then + triage_log "wedge window-scoped marker write FAILED AND rollback AND sentinel write ALL FAILED on $win - no queue-flood suppression (sentinel write itself failed under the same fs failure); operator MUST intervene immediately to resolve the fs condition; every retry will re-fire PERMANENTLY-WEDGED" + exit 3 + fi + if [ "$_rm_status" -ne 0 ]; then + triage_log "wedge window-scoped marker write FAILED on $win AND rollback ALSO FAILED - sentinel .wedge-rollback-failed-$key set (TTL $FM_ROLLBACK_SENTINEL_TTL_SECS); no wake-amplification, operator must intervene (rm both .wedge-permanent- and .wedge-rollback-failed-$key)" + exit 2 + fi + triage_log "wedge window-scoped marker write FAILED: $STATE/.wedge-permanent-$(window_key "$win") - rolled back per-hash marker AND cap state (escalation counter and stale timer reset); operator must intervene" + exit 1 + fi + rm -f "$since_file" + clear_write_tracking "$(window_key "$win")" + # v15: successful cap-fire clears any prior rollback-failed + # sentinel - the fs condition that caused the prior sentinel has + # been resolved. + rm -f "$STATE/.wedge-rollback-failed-$(window_key "$win")" + # v11 (2026-09-07, Greptile P1): reset the wedge-escalation counter + # when the cap fires. Without this, a subsequent fresh-hash poll + # (busy pane churning its elapsed-time footer, pane re-rendering + # for any other reason) reads n=$(( $(cat $escalation_file) + 1 )) + # = $max + 1 = saturated, fires PERMANENTLY-WEDGED immediately on + # the new hash, and continues to fire on every subsequent hash. The + # per-hash marker for the OLD hash is still in place so the SAME + # hash is silenced by the early-return at the top of this function; + # a NEW hash needs the counter fresh so its wedge escalates + # independently and re-engages the cap on its own merits. + : > "$escalation_file" + triage_log "wedge permanently capped: $win (escalation $n, max $FM_WEDGE_MAX_ESCALATIONS, hash ${hash:0:12})" + wake "$reason" + return 0 + fi # v20: end cap-fire block (also closes FM_WEDGE_MAX_ESCALATIONS>0 guard) fm_wake_append stale "$win" "$reason" || exit 1 rm -f "$since_file" clear_write_tracking "$(window_key "$win")" @@ -1557,6 +1762,16 @@ handle_paused_stale() { # key=$(window_key "$win") printf '%s' "$h" > "$STATE/.stale-$key" : > "$STATE/.paused-$key" + # LOCAL PATCH (2026-08-19, v5): do NOT clear .wedge-permanent--* here. + # A pause-class transition is NOT proof that the underlying wedge has + # resolved - the operator may have declared `paused:` precisely because the + # wedge was unfixable in real time. Clearing the permanent marker would + # re-arm the cap, so when the pause lifts the same still-wedged hash would + # climb back to FM_WEDGE_MAX_ESCALATIONS and fire another terminal wake. + # The marker is keyed on (window, hash) so it is naturally stale if the + # wedge genuinely resolves (next poll sees a new hash, fresh cap cycle). + # Manual operator reset (e.g. `rm STATE/.wedge-permanent--H12`) is the + # only legitimate way to lift the cap. rm -f "$STATE/.stale-since-$key" "$STATE/.wedge-escalations-$key" clear_write_tracking "$key" statusf="$STATE/$task.status" @@ -1564,7 +1779,7 @@ handle_paused_stale() { # case "$mtime" in ''|*[!0-9]*) mtime=$(date +%s) ;; esac now=$(date +%s) age=$(( now - mtime )) - last=$(status_declared_wait_line "$statusf") + last=$(last_status_line "$statusf") min_age=$PAUSE_RESURFACE_SECS declaration="declared:$(fm_wake_signal_sig "$statusf" || true)" if status_is_captain_held "$last"; then @@ -1622,7 +1837,7 @@ handle_paused_stale() { # busy_turn_bound_check() { # local win=$1 task=$2 h=$3 since_file=$4 escalation_file=$5 key statusf declared statusf="$STATE/$task.status" - if status_is_paused_or_captain_held "$(status_declared_wait_line "$statusf")"; then + if status_is_paused_or_captain_held "$(last_status_line "$statusf")"; then if afk_present; then # Away mode is daemon-owned, so this bound hands off the PLAIN wake identity # and lets the daemon classify the declaration itself - the undecorated @@ -1647,7 +1862,7 @@ busy_turn_bound_check() { # "$STATE/.stale-$key" triage_log "absorbed busy over-age pane (captain-held, never rechecked while the away-posture record exists): $win" return 0 @@ -1687,6 +1902,13 @@ clear_pause_tracking() { # local key=$1 clear_pause_state "$key" clear_stale_hash_tracking "$key" + # LOCAL PATCH (2026-08-19, v5): do NOT clear .wedge-permanent--* here. + # A full pause-tracking reset is not the same as the wedge genuinely + # resolving. The hash will change on the next stale poll, at which point the + # marker for the old hash is naturally stale clutter (no fresh + # wedge_timer_check call would ever look up that marker again - the lookup + # key is the new hash). Manual operator action is the only legitimate way + # to lift the cap. } # Reconcile a declared pause or captain-held status with authoritative crew state. @@ -1696,7 +1918,7 @@ clear_pause_tracking() { # pause_state_class() { # local win=$1 task=$2 key last recheck_file class agent_alive kind key=$(window_key "$win") - last=$(status_declared_wait_line "$STATE/$task.status") + last=$(last_status_line "$STATE/$task.status") recheck_file="$STATE/.paused-rechecked-$key" if ! status_is_paused_or_captain_held "$last"; then rm -f "$recheck_file" @@ -1869,7 +2091,7 @@ surface_nonterminal_stale() { # local win=$1 h=$2 key task last declared=1 bounded=1 throttled=1 until now key=$(window_key "$win") task=$(window_to_task "$win" "$STATE") - last=$(status_declared_wait_line "$STATE/$task.status") + last=$(last_status_line "$STATE/$task.status") STALE_WAIT_DECLARATION= if status_is_paused "$last"; then declared=0 @@ -2345,40 +2567,12 @@ if ! fm_procevent_launch_confirm_seconds >/dev/null; then exit 1 fi -# evict_stalled_holder : retire a live lock holder whose beacon stalled past -# WATCHER_STALL_BOUND. The pid is signalled only while it still proves the -# lock's own recorded identity (fm_watcher_lock_matches_pid: this home, this -# script, and the starttime+cmdline proof the lock carries), so a recycled pid -# is never touched; TERM only, never KILL, and never a name or pattern match. -# Succeeds only once the holder has exited within the bounded wait. -evict_stalled_holder() { - local pid=$1 i=0 - fm_watcher_lock_matches_pid "$STATE" "$WATCH_PATH" "$pid" "$FM_HOME" || return 1 - kill -TERM "$pid" 2>/dev/null || return 1 - while [ "$i" -lt 50 ] && fm_pid_alive "$pid"; do - sleep 0.1 - i=$((i + 1)) - done - ! fm_pid_alive "$pid" -} - -EVICTED_PID= -EVICTED_BEAT_AGE= -BEAT="$STATE/.last-watcher-beat" -while ! fm_lock_try_acquire "$WATCH_LOCK"; do +if ! fm_lock_try_acquire "$WATCH_LOCK"; then + BEAT="$STATE/.last-watcher-beat" if [ -n "${FM_LOCK_HELD_PID:-}" ]; then if [ -e "$BEAT" ]; then beat_age=$(fm_path_age "$BEAT") if [ "$beat_age" -ge "$WATCHER_STALE_GRACE" ]; then - # One eviction per arm: the retry re-reads the lock and beacon, so a - # holder that exited leaves a dead-pid lock the normal reclaim takes, - # and a rival arm that won first reads as a fresh running watcher. - if [ -z "$EVICTED_PID" ] && [ "$beat_age" -ge "$WATCHER_STALL_BOUND" ] \ - && evict_stalled_holder "$FM_LOCK_HELD_PID"; then - EVICTED_PID=$FM_LOCK_HELD_PID - EVICTED_BEAT_AGE=$beat_age - continue - fi echo "watcher: lock held by live pid $FM_LOCK_HELD_PID but heartbeat is stale for ${beat_age}s (>${WATCHER_STALE_GRACE}s); inspect or stop that watcher before re-arming." >&2 exit 1 fi @@ -2391,9 +2585,6 @@ while ! fm_lock_try_acquire "$WATCH_LOCK"; do echo "watcher: already running" fi exit 0 -done -if [ -n "$EVICTED_PID" ]; then - echo "watcher: replaced stalled pid $EVICTED_PID (beacon ${EVICTED_BEAT_AGE}s past hard bound ${WATCHER_STALL_BOUND}s)" fi WATCHER_RECOVERY_PENDING=0 if [ -n "${FM_LOCK_RECOVERED_PID:-}" ]; then @@ -2583,29 +2774,6 @@ resurface_after_downtime() { } while :; do - # Home-gone exit: a deleted home, state directory, or code root means this - # watcher's world is gone (a torn-down temporary home or a discarded - # disposable checkout). Exit with a logged reason rather than writing state - # into nothing, or into a live home from a checkout that no longer exists. - # A detached helper this watcher started (home-summary refresh, reconcile) - # can recreate a deleted state directory before the next poll, so a lock - # with no holder at all is read as the same teardown: only a fresh watcher - # ever recreates the lock, and that case is the self-eviction below. - # Scoped to this process alone: no other watcher is signalled. - if [ "$WATCH_HOME_EXISTED" -eq 1 ] && [ ! -d "$FM_HOME" ]; then - echo "watcher: exiting - home no longer exists: $FM_HOME" >&2 - exit 1 - elif [ ! -d "$STATE" ]; then - echo "watcher: exiting - state directory no longer exists: $STATE" >&2 - exit 1 - elif [ ! -e "$WATCH_LOCK/pid" ]; then - echo "watcher: exiting - state directory was torn down (singleton lock removed): $STATE" >&2 - exit 1 - elif [ ! -d "$SCRIPT_DIR" ]; then - echo "watcher: exiting - code root no longer exists: $SCRIPT_DIR" >&2 - exit 1 - fi - # Self-eviction: if the singleton lock no longer names this process, a second # watcher has taken over (e.g. a transient duplicate from a racy arm). Stand # down so the rightful singleton continues alone. The EXIT trap's release @@ -2641,16 +2809,6 @@ while :; do # No conversation scraping; unresolved records are never silently expired. fm_pending_reply_tick "$STATE" || true - # Endpoint liveness runs before queue observation: a positively dead or - # missing secondmate endpoint is relaunched here on a bounded cadence, which - # is also what unsticks that mate's foreign wake queue. The tick's single - # wake exits the cycle like every other wake, so its marker is stamped before - # any relaunch and the restarted watcher will not re-probe early. - secondmate_liveness_tick || { - echo "watcher: secondmate liveness check failed" >&2 - exit 1 - } - # A live secondmate endpoint does not prove that its own wake loop is alive. # Observe the foreign queue before the rest of this cycle so an aged row wakes # the parent without consuming or rewriting the receiving home's record. @@ -2763,17 +2921,6 @@ EOF fi reason="check: $c: $out" if [ "$is_pr_poll" -eq 1 ] && [ "$out" = merged ]; then - if [ "$(fm_meta_get "$STATE/$id.meta" kind)" = secondmate ]; then - # A merge poll armed on a secondmate is residue: the mate is a - # persistent worker, never landed work, and the merge it detected - # belongs to a task in the mate's own home. Retire the poll with no - # outcome and no wake; bin/fm-pr-check.sh refuses to arm another. - retire_merged_pr_poll "$id" - pr_poll_control_release || exit 1 - touch "$STATE/.last-check" - triage_log "retired a merge poll armed on secondmate $id without reporting an outcome" - continue - fi if ! fm_merge_authority_read "$STATE" "$id" \ "$provider" "$host" "$path" "$number"; then triage_log "no matching persisted merge authority for $id; recording an external merge outcome" @@ -2959,7 +3106,7 @@ EOF # exemption below, because a mate's steers land in an inbox too. [ -z "$task" ] || inbox_steer_check "$w" "$task" key=$(window_key "$w") - last=$(status_declared_wait_line "$STATE/$task.status") + last=$(last_status_line "$STATE/$task.status") if ! status_is_paused_or_captain_held "$last" && [ -e "$STATE/.paused-$key" ]; then clear_pause_tracking "$key" fi @@ -3088,6 +3235,16 @@ EOF task=$(window_to_task "$w" "$STATE") case "$(pause_state_class "$w" "$task")" in working) + # v9 (2026-08-25): the cap marker is keyed on (window, hash) and + # bounded by FM_CAP_HORIZON_SECS (the marker file's timestamp is + # checked at the top of wedge_timer_check). No auto-lift on + # recovery is needed - a stale cap expires after the horizon + # (v12: the window-scoped marker silences fresh hashes too, so + # a hash change alone does not re-engage before expiry). + # pause_state_class=working can be a steady state + # during a wedge (the worker is doing things but the pane is + # static), so it is NOT a recovery signal - the v6/v7 lift + # sites on this verdict over-corrected and let the cap cycle. clear_pause_tracking "$key" printf '%s' "$h" > "$sf" date +%s > "$ssf" @@ -3102,9 +3259,11 @@ EOF esac else task=$(window_to_task "$w" "$STATE") - if [ -e "$pf" ] || status_is_paused_or_captain_held "$(status_declared_wait_line "$STATE/$task.status")"; then + if [ -e "$pf" ] || status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")"; then case "$(pause_state_class "$w" "$task")" in paused) handle_paused_stale "$w" "$task" "$h" ;; + # v9: same-hash + was-paused + working pipeline. Cap is horizon- + # bounded, no auto-lift here. working) clear_pause_state "$key" printf '%s' "$h" > "$sf" wedge_timer_check "$w" "$ssf" "non-terminal stale (provably working after a declared pause)" "$ewf" "$task" "$h" @@ -3112,6 +3271,10 @@ EOF *) handle_paused_stale "$w" "$task" "$h" ;; esac else + # v9: same-hash branch with no declared pause. Cap is horizon- + # bounded (FM_CAP_HORIZON_SECS), no explicit lift. The v6/v7 + # attempts at "lift on pause_state_class=working" were over- + # eager (the verdict can be steady-state during the wedge). wedge_timer_check "$w" "$ssf" "non-terminal stale" "$ewf" "$task" "$h" fi fi @@ -3132,7 +3295,7 @@ EOF # is cleared - but not in the same poll the declared-pause cadence just # recorded it, or the re-surface throttle it depends on would be erased and # the pause would re-surface every poll instead of once per long cadence. - if [ "$paused_bound" -ne 0 ] && [ -e "$pf" ] && { [ "$n" -ge 2 ] || ! status_is_paused_or_captain_held "$(status_declared_wait_line "$STATE/$(window_to_task "$w" "$STATE").status")"; }; then + if [ "$paused_bound" -ne 0 ] && [ -e "$pf" ] && { [ "$n" -ge 2 ] || ! status_is_paused_or_captain_held "$(last_status_line "$STATE/$(window_to_task "$w" "$STATE").status")"; }; then clear_pause_tracking "$key" fi fi @@ -3140,14 +3303,27 @@ EOF printf '%s' "$h" > "$hf" echo 0 > "$cf" paused_bound=1 + # The busy-turn wedge timer is deliberately NOT reset on a hash change: a + # genuinely wedged worker (Pi's ticking elapsed-time footer, the original + # incident) renders a new hash every poll, so clearing the timer here + # would restart it forever and no busy-turn wedge could ever escalate. + # Only the non-busy-bound path clears the pending escalation bookkeeping. if [ "$busy_now" -eq 0 ] && busy_turn_over_age "$task"; then busy_turn_bound_check "$w" "$task" "$h" "$ssf" "$ewf" && paused_bound=0 else rm -f "$ssf" "$ewf" clear_write_tracking "$key" fi + # v12 (2026-09-07): NO auto-lift of the window-scoped marker here. + # The marker is the primary gate against the hash-churning busy-pane + # loop Greptile flagged; lifting it on every hash-change idle verdict + # would re-introduce the v6/v7/v8 over-eager recovery the v9 design + # removed (a pane can be classified idle on a hash change and still + # be the same underlying wedge). The window-scoped marker is lifted + # only by FM_CAP_HORIZON_SECS elapsing or an operator removing BOTH + # markers - no code path in this script clears it on its own. task=$(window_to_task "$w" "$STATE") - if ! afk_present && status_is_paused_or_captain_held "$(status_declared_wait_line "$STATE/$task.status")" && [ "$busy_now" -ne 0 ]; then + if ! afk_present && status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")" && [ "$busy_now" -ne 0 ]; then case "$(pause_state_class "$w" "$task")" in paused) handle_paused_stale "$w" "$task" "$h" ;; # Inconclusive, but the declared wait itself still stands, so only the diff --git a/docs/architecture.md b/docs/architecture.md index fa94a7349f2..d8613de74e4 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -17,6 +17,8 @@ The throttle is scoped to both the current captain-call lifecycle and the status A secondmate reaches the stale path only for a wait declared in its status line, so a hold recorded only in the backlog while its last line is `working:` or `done:` is outside this guard. Reaching that case would require consulting the backlog for windows the secondmate gate deliberately skips, putting backlog reads on the ordinary poll hot path this design preserves. Repeated provably-working stale escalations on the same unchanged pane add an escalation count to the wake reason and, at `FM_WEDGE_DEMAND_INSPECT_COUNT`, a `demand-deep-inspection` marker. +Past `FM_WEDGE_MAX_ESCALATIONS` (default 0 = cap disabled; captains opt in by setting `FM_WEDGE_MAX_ESCALATIONS=N` with `N>=1`) consecutive wedge escalations on the same window, the watcher emits one terminal `PERMANENTLY-WEDGED` wake and writes BOTH `STATE/.wedge-permanent-` (window-scoped, silences all hashes in the window) and `STATE/.wedge-permanent--` (per-hash); subsequent polls for any hash in that window short-circuit until `FM_CAP_HORIZON_SECS` elapses or the operator manually removes both markers. The window-scoped marker prevents a hash-churning pane from rebuilding the escalation counter per fresh hash. v20 (2026-09-24) flip: unconfigured path leaves the cap off and preserves the pre-PR behavior, addressing the VISION.md "Authority is explicit" concern kunchenguid's firstmate raised on PR #2605. +`tests/fm-watch-wedge-cap.test.sh` pins the cap-fire ordering and both marker writes, silence across pause transitions and fresh hashes until the horizon, invalid-override fallbacks, and the rollback-failed sentinel short-circuit. In the same branch that is about to escalate, the pane's own account of its quiet is consulted first: the worker's declared `paused:` or verified `captain-held` status line. That declaration defers the escalation to the `FM_PAUSE_RESURFACE_SECS` recheck cadence instead, because a lane waiting on something it named is silent for a reason the escalation would misreport, and the ladder would otherwise climb for as long as the wait lasts. A declared clearing time (`paused: ... until `) that has already passed stops counting as that account, so a lane whose own wait is over, and a lane that never declared one, both keep the unchanged escalation schedule, reason and `demand-deep-inspection` wording. @@ -66,13 +68,9 @@ Agent endpoint liveness and queue-consumption liveness are separate: on each pol A queue that is draining is not stalled, so the primary times the interval since that oldest actionable row last changed rather than the age of the row itself, and rows that declare themselves a bounded external wait (`awaiting external - declared pause`) are not actionable evidence at all. Once that no-progress interval reaches `FM_SECONDMATE_WAKE_STALL_SECS` and the mate is not provably inside an active turn (an exact busy verdict, honored only while that same no-progress interval is under `FM_BUSY_TURN_MAX_SECS`, because a mate's turns end in its own home and leave no completed-turn evidence in the primary's), a mate whose semantic busy class is exactly idle, whose agent is alive, and whose composer is not pending is rung once so its own home can drain, and the parent notification is withheld until that same row stays frozen for another stall interval; unknown, busy-over-bound, and ring-unsafe panes keep the parent alarm, and empty inbox or a fresh child beacon is not idle proof. The primary then appends one keyed `check` wake naming the mate, row sequence, and observed idle interval; parent receipts and queued-key deduplication suppress repeats across watcher and handling crashes, one notification covers a whole no-progress episode, and any move of that position - drain progress, or the fresh rows of a queue reprovisioned under the same task id, at whatever sequence it restarts - ends that episode and starts a fresh observation interval, while empty, advancing, and declared-wait queues remain silent. -Endpointless registered mates remain outside this queue scan because its preconditions can never be met for them. -Dead-or-missing endpoint recovery is instead shared by two drivers over one library, `bin/fm-secondmate-liveness-lib.sh`: the session-start sweep in `bin/fm-bootstrap.sh`, and the watcher's own `FM_SECONDMATE_LIVENESS_SECS`-cadence tick during ordinary supervision. -Both relaunch only the recovery-grade `dead` and `missing` verdicts through the ordinary guarded `fm-spawn.sh --secondmate` path, a remote route is probed read-only across its host-local boundary and is never replaced by a local endpoint, and the per-mate liveness lock keeps a concurrent sweep and tick from killing or re-probing an endpoint the other is mid-relaunch on. -Each automatic relaunch surfaces as exactly one `check` wake plus a durable line in `state/.secondmate-relaunch-`, and a mate that exceeds `FM_SECONDMATE_LIVENESS_MAX_ATTEMPTS` ledgered attempts inside `FM_SECONDMATE_LIVENESS_WINDOW_SECS` is parked behind a bound marker and escalated once until a live probe rearms it with a full attempt budget. +Endpointless registered mates remain outside this scan because startup secondmate-liveness owns dead or missing endpoint recovery, and remote homes retain their host-local supervision boundary. `tests/fm-wake-queue.test.sh` pins the no-progress notification, drain-progress reset, declared-pause exclusion, active-turn deferral, proven-idle child-first ring, busy and unknown parent-alarm paths, genuine stall after a ring, idempotence, quiet-queue, and byte-for-byte foreign-row preservation guarantees. -When a canonical validated task PR poll returns exactly `merged`, the watcher routes it through the shared merge-outcome emitter before retiring the poll. -A legacy poll armed on a persistent `kind=secondmate` record is residue from a child's relayed PR: the watcher retires it without a merge outcome, notification marker, or wake, leaving the mate's lifecycle intact; `bin/fm-pr-check.sh` refuses new polls on such records. +When a canonical validated PR poll returns exactly `merged`, the watcher routes it through the shared merge-outcome emitter before retiring the poll. [`bin/fm-merge-outcome-lib.sh`](../bin/fm-merge-outcome-lib.sh)'s header owns role routing, PR-specific wake identity, marker-locked normal deduplication, and the at-least-once ordering that prefers a rare duplicate over silence. After successful outcome publication, the watcher immediately delivers the emitter's local actionable poll row and publishes a private retirement receipt bound to the poll's registration, bytes, file identities, metadata, provider, URL, and task ID. The retirement receipt makes poll cleanup safely retryable across restarts: fixed-path recovery revalidates the same evidence, removes the runnable check first, removes its registration and data sidecars, removes the receipt last, and preserves task metadata including `pr=` and `pr_head=`. @@ -92,7 +90,7 @@ A crew that declares `paused:` for a known external wait, or carries a verified For an ordinary crew that has stopped, the normal-mode watcher first surfaces one stale wake, then applies that same cadence to an unchanged `paused:` or durable `captain-held` endpoint while attended; the pause classification itself is recovered only when the backend confidently reports its agent dead. Live or inconclusive liveness remains fail-open at that initial surface, so a worker genuinely waiting on a decision is never silenced. Its later sights are still held to that same bounded cadence rather than re-alarming on every pane-hash change, because the throttle is keyed to the declaration and not to the pane an idle parked worker keeps ticking. -The pause path still never reads a secondmate's endpoint liveness - dead-or-missing recovery belongs to the dedicated liveness tick above - and a mate is admitted to that same cadence only to serve a status-declared wait's bounded re-surface, so a forgotten `paused:` declaration, or an attended `captain-held` declaration, cannot rot invisibly. +A secondmate's endpoint liveness is still never read at all; a mate is admitted to that same cadence only to serve a status-declared wait's bounded re-surface, so a forgotten `paused:` declaration, or an attended `captain-held` declaration, cannot rot invisibly. Its initial normal-mode status signal still surfaces through the no-verb path, while a daemon-backed away posture self-handles that routine signal and owns later external-wait rechecks. Fresh stale panes use the same current-state read before trusting the status log, so an active run or a proven busy worker outranks an old captain-relevant status-log line left behind before validation. No-change heartbeats are also benign. @@ -127,7 +125,7 @@ The most recent recognized ci log marker wins, so checks-green monitoring report `bin/fm-crew-state.sh` owns the evidence guard that recognizes ended CI monitors after green checks, including cancelled runs and skipped rebase steps; a passed run alone never proves a forge merge. In the coarse runs-ledger fallback, which has no steps table and no ci log, a terminal failed record whose daemon an explicit `daemon status` probe proves down reports unknown as unverified instead: an instrument failure must never read as work failure. The same instrument rule covers the ledger-anchored continuation of a selected run whose head this copy cannot resolve: once the probe answers down, that still-executing record reports unknown as unverified, while a run parked at a gate keeps its gate and findings because an open decision stays open when the instrument dies, and a `needs-decision` or `blocked` event the crew observed first hand stays open with the unverified record named as the reason rather than superseded by it. -Only when no matching run exists does it consult semantic busy state; exact busy reports working, exact idle permits fallback to the log's resolved current declaration - the newest decision the fold still holds open, otherwise a declared wait still standing after later resolved lines for other keys, otherwise the latest recognized event - when its verb maps to a recognized run-state, and unknown or a dead pane stays unknown instead of trusting a stale log. +Only when no matching run exists does it consult semantic busy state; exact busy reports working, exact idle permits fallback to the log's resolved current declaration - the newest decision the fold still holds open, otherwise the latest recognized event - when its verb maps to a recognized run-state, and unknown or a dead pane stays unknown instead of trusting a stale log. Decision-only events such as `resolved` never become current state or leak their prose into the current-state detail. In that status-log fallback, a declared external wait reports the distinct `paused` state with its reason. The semantic branch reports working only on an exact busy verdict and names the source that produced it; an unknown verdict never becomes working, never permits the status-log fallback, and never becomes a silent idle. @@ -144,8 +142,7 @@ The script header owns the exact JSON schema. On a Pi primary, supervision is default-on: the watcher extension can hand eligible task-local rows from an ordinary actionable wake, plus selected fleet-wide heartbeat reviews, to a persistent in-process supervision conversation while main-only rows remain on the captain-facing path. The branch handles those rows, stores the outcome durably, and merges it back into main. A captain-facing outcome persists as one exact, sequence-keyed visible transcript entry and then opens one sequence-keyed processing turn on main, which only main's sequence-bound acknowledgement closes. -[docs/pi-supervision-branch.md](pi-supervision-branch.md) owns row eligibility, dispatch architecture, deterministic outcome delivery, and processing re-presentation, while the generated [Pi supervision protocol](supervision-protocols/pi.md) owns MAIN's merged-event handling and acknowledgement duty. -For the opt-in away-posture exception to the non-Pi harnesses' wake-to-main path, see [supervision-host.md](supervision-host.md). +[docs/pi-supervision-branch.md](pi-supervision-branch.md) owns row eligibility, dispatch architecture, deterministic outcome delivery, and processing re-presentation, while the generated [Pi supervision protocol](supervision-protocols/pi.md) owns MAIN's merged-event handling and acknowledgement duty; every other harness keeps the wake-to-main path unchanged. ### Registered secondmate current state @@ -167,7 +164,7 @@ That block owns the live wait shape for the running primary harness: Claude's St The arm layer records one bounded lifecycle row per observed cycle in `state/.watch-cycle-exits.log`; `state/.watch-triage.log` remains exclusively the absorbed-wake debug log. Pi, omp, and OpenCode verify session-lock ownership and launch one singleton successor from their child-close handlers before delivering an actionable wake prompt, with bounded exponential retry for failed restoration. Pi additionally retains an established predecessor across ordinary same-process session shutdown until the replacement generation commits its tracked arm, and its active-versus-handoff generation marker prevents an absent replacement extension from satisfying the fresh-beacon handoff tolerance. -Claude's `bin/fm-claude-stop-autoarm.sh` hook fires on every Stop and, when the home is eligible and still needs supervision, claims one home-scoped cycle, foregrounds the arm wrapper (or the [opt-in supervision host](supervision-host.md)), and translates actionable closes into exit-2 rewakes. +Claude's `bin/fm-claude-stop-autoarm.sh` hook fires on every Stop and, when the home is eligible and still needs supervision, claims one home-scoped cycle, foregrounds the arm wrapper, and translates actionable closes into exit-2 rewakes. It suppresses failed-looking closes when the same identity-matched watcher is healthy, retries genuine failures within a bound, and coordinates exhausted failure episodes with the Claude turn-end guard as documented in [`turnend-guard.md`](turnend-guard.md). [`watcher-continuity.md`](watcher-continuity.md) owns Claude's residual active-turn coverage and watcher-status command-gating boundary. Cursor's `bin/fm-turnend-guard-cursor.sh` hook is the same between-turns shape in one synchronous step: it parks the awaited `stop` hook on the arm wrapper and translates an actionable close into one `followup_message`, with a generation baton that makes an older park still running after the next `stop` claim stand down instead of leaking a stale duplicate wake. @@ -185,15 +182,13 @@ Away mode is a posture of the one supervision session, recorded in `state/.afk-c The captain's away words are the whole mandate: the record owner's header is the single owner of the record schema, the words are recorded verbatim, and by the captain's mandate no parser, tokenizer, classifier, or grammar reads them anywhere. The supervision session reads the words at the tail of every wake and acts on them by its own judgment at the moment an event makes them relevant, only through the guarded scripts under standing authority, never by analogy, holding for the return on doubt; `bin/fm-branch-prompt.sh` "Postures" owns those execution rules. What stays mechanical is exactly what a script can check without reading words: a merge green at its live head under the record lock, synchronous merges only, the spend cap, and the never-set; destructive, irreversible, and security-sensitive actions are never pre-authorizable whatever the words say. -The record's presence is the posture on every harness, `bin/fm-afk-launch.sh` owns entry and exit, and `bin/fm-afk-return.sh` archives the record and owns the return brief's ordered sections, including landed live task records that still owe cleanup, rendered from durable state; persistent secondmates are excluded from that cleanup section even if an older record carries a child's merged PR. +The record's presence is the posture on every harness, `bin/fm-afk-launch.sh` owns entry and exit, and `bin/fm-afk-return.sh` archives the record and owns the return brief's ordered sections, including landed live task records that still owe cleanup, rendered from durable state. While the record exists neither supervisor rechecks an item held for the captain, and a declared external wait names when it clears with `until` for a condition-aware recheck in both postures that occurs at the declared time or the hours-long `FM_PAUSE_RESURFACE_SECS` bound, whichever comes first. On Pi and pi-signed the away daemon is no longer launched: the ordinary supervision session continues under the record with main parked, so the supervision branch takes every actionable wake, captain outcomes accumulate for the return brief, and main's standing authority relocates to the branch through the guarded scripts, each keeping its own gate ([`pi-supervision-branch.md`](pi-supervision-branch.md#postures)); a wake the branch cannot take and a watcher failure still reach main. -On an opted-in non-Pi home, the [supervision host](supervision-host.md) runs the away session instead of the daemon. -A presence-gated sub-supervisor (`bin/fm-supervise-daemon.sh`) still extends walk-away supervision on the remaining harnesses: the `/afk` skill starts it through the tracked foreground helper `bin/fm-afk-start.sh` once the record exists, after which the watcher reverts to daemon-managed one-shot mode and the daemon self-handles routine wakes in bash. +A presence-gated sub-supervisor (`bin/fm-supervise-daemon.sh`) still extends this for walk-away supervision on the other harnesses: the `/afk` skill starts it through the tracked foreground helper `bin/fm-afk-start.sh` once the record exists, after which the watcher reverts to daemon-managed one-shot mode and the daemon self-handles routine wakes in bash. The watcher and daemon share `bin/fm-classify-lib.sh` for captain-relevant status verbs, declared-wait vocabulary (a `paused:` external wait and a verified `captain-held` transfer alike, through one combined predicate), and status-scan primitives. Terminal verbs remain captain-relevant, while a nonterminal progress verb cannot become terminal merely because its prose contains a legacy free-text token such as `merged`; bare legacy free-text lines remain compatible. -The shared latest-event read takes the most recent line that leads with a recognized verb, a legacy token, or an unrecognized status prefix, so a bad declaration stays visible as itself while continuation prose and trailing blank lines after a multi-line record cannot hide a declared wait. -Both supervisors decide a declared wait through the library's declared-wait read rather than that latest event, so a later `resolved` line for a different phase key - including an `fm-send --resolve-key default` answer to a keyless decision - does not end a standing keyless or keyed `paused:` wait, while a resolved line for the wait's own key or any other later event still does. +The shared latest-event read takes the most recent line that leads with a recognized verb or legacy token, so continuation prose and trailing blank lines after a multi-line record cannot hide a declared wait. Both supervisors classify the status bytes appended since they last classified that log, never its last line alone, and report every actionable event through the captured endpoint before committing that position. The watcher's `.seen-*` and `.hb-surfaced-` markers and the daemon's `.subsuper-seen-status-` marker independently track reported file state and successfully classified position, so an unchanged unreadable state reports once without advancing past unread content, while a changed state retries and an unusable position re-reads the whole log. A keyed `needs-decision` or `blocked` transition accepted by the whole-file decision fold is retired only when that fold retires it - an explicit close for its exact key, or a terminal declaration by the ship or scout that owns the log - while a reserved-key transition the fold rejects surfaces as a reconciliation signal without becoming an open decision. @@ -259,7 +254,7 @@ Herdr's native `agent.get` verdict still participates, but only as evidence of a tmux, zellij, orca, and cmux expose no native busy primitive at all, so a task on those backends is classified purely from its adapter's own lifecycle record. That poll loop is still the default event source for backends with no native push events, so this stays an extraction of the abstraction rather than a watcher rewrite. For capable Herdr sessions, the same watcher replaces its terminal sleep with a bounded native event wait that immediately surfaces `blocked`; [Push events and polling fallback](herdr-backend.md#push-events-and-polling-fallback) owns the current mechanism and capability gates, while [runtime backend verification](verification/runtime-backends.md#native-blocked-event) owns the active evidence. -The deeper agent-process liveness probe is separate from that busy-state poll and is shared by the session-start sweep and the watcher's liveness tick through `bin/fm-secondmate-liveness-lib.sh`: tmux and Herdr have verified classifiers for secondmate recovery, Zellij remains unverified, and Orca and cmux do not support secondmate spawns. +The deeper session-start agent-process liveness probe is separate from that busy-state poll: tmux and Herdr have verified classifiers for secondmate recovery, Zellij remains unverified, and Orca and cmux do not support secondmate spawns. Herdr can be selected explicitly or by runtime auto-detection: Treehouse remains its worktree provider, [`herdr-backend.md`](herdr-backend.md) owns current setup, CI coverage, and safety limits, and [`verification/runtime-backends.md`](verification/runtime-backends.md#herdr) owns active empirical evidence. Herdr uses one tab per task; [Watching and task containers](herdr-backend.md#watching-and-task-containers) owns launcher-bound workspace placement, the label-only fallback, and recovery scope. Its default-on presentation projection may place one clean new task in a disposable workspace without changing endpoint authority or lifecycle ownership; [Presentation spaces](herdr-backend.md#presentation-spaces) owns that conditional design, the Herdr version floor its unconfigured default is gated behind, and its narrow home-local restored-shell cleanup at locked session start. @@ -285,16 +280,16 @@ Only a named non-default branch checked out in `FM_ROOT` is a worktree tangle. `fm-tangle-lib.sh` resolves the default branch from `origin/HEAD`, then local `main` or `master`, and classifies that named non-default primary branch as the tangle. `fm-guard.sh` prints the repair command on the next mutable fleet action, while `bin/fm-session-start.sh` reports the same condition through bootstrap as a `TANGLE:` line at session start. If another live session holds the fleet lock, both surfaces keep the alarm but switch to read-only wording with no repair command. -Ship briefs also tell the crewmate to verify `pwd -P` and `git rev-parse --show-toplevel` before creating its ship branch (`fm/` by default, or the project's registered prefix), then stop with a blocked status if it landed in the primary checkout. +Ship briefs also tell the crewmate to verify `pwd -P` and `git rev-parse --show-toplevel` before creating `fm/`, then stop with a blocked status if it landed in the primary checkout. Placement is proven only at launch, so `bin/fm-spawn.sh` also exports the task id as `FM_TASK_ID` into every ship and scout pane, and `bin/fm-test-run.sh` refuses to execute the behavior suite from the primary checkout while that marker is set; the runner's header owns the predicate and [`tests/fm-test-run.test.sh`](../tests/fm-test-run.test.sh) pins it. ## No-mistakes gate authority boundary Firstmate's own no-mistakes gate runs agents inside a checkout that also contains the fleet-captain identity in `AGENTS.md`, so gate execution needs an authority boundary separate from ordinary crewmate worktree isolation. The tracked `.no-mistakes.yaml` sets `disable_project_settings: true`; no-mistakes honors that setting only from the trusted default-branch copy, so a pushed branch cannot enable its own project instructions during validation. -Independently, the fleet lifecycle entrypoints use `bin/fm-gate-refuse-lib.sh` to refuse gate calls against the real fleet, while permitting validation against a disposable lab home minted by `bin/fm-lab-home.sh`. -A normal primary checkout or crewmate worktree remains unaffected. -The refusal library's header owns the gate detection, lab-home exception, test-harness bypass, and relationship to no-mistakes' HEAD-continuity guard; the lab helper's header owns its usage. +Independently, `fm-spawn.sh`, `fm-send.sh`, `fm-control.sh`, and `fm-teardown.sh` source `bin/fm-gate-refuse-lib.sh` and exit with status 3 before fleet mutation when the gate environment marker is present or the current checkout matches the default no-mistakes gate-repository topology. +A normal primary checkout or crewmate worktree has neither signal and remains unaffected. +The helper's header owns the exact signal detection, relocated-home limitation, test-harness bypass, and relationship to no-mistakes' HEAD-continuity guard. ## Two task shapes @@ -370,17 +365,16 @@ On a `forge=gerrit` project both `no-mistakes` and `direct-PR` end with the work Firstmate passes the binding unchanged to `bin/fm-brief.sh --forge` and never infers one from a remote, host, or protocol; a ship spawn reads it from the registry through `bin/fm-project-mode.sh --forge` and refuses a brief that disagrees with it, and a promotion reads it the same way for the binding alone. `bin/fm-forge-detect.sh` only proposes a binding at project-add intake; nothing re-derives one from a clone at use time. `bin/fm-project-mode.sh` remains the one registry parser for the mechanical consumers that have no task in hand: fleet sync's `local-only` skip and home seeding's refusal and no-mistakes initialization. -The registry's optional `branch=` annotation overrides a project's ship-branch prefix (default `fm/`) the same way: firstmate resolves it via `bin/fm-project-mode.sh --branch-prefix` at intake and passes it explicitly to `bin/fm-brief.sh --branch-prefix`, which never reads the registry itself; each script's own header owns its side of that contract. When a selected delivery path calls for a diff, `bin/fm-review-diff.sh` refreshes the authoritative base and, when task meta records a GitHub pull-request `pr=`, always fetches and compares against `refs/pull//head` by default (recorded `pr_head=` is only an offline fallback) before falling back to the local branch with a warning. A GitLab merge request and a Gerrit change expose no such ref, so a task recording one of those diffs the local branch under that same warning, which is its current content. Where a no-mistakes pipeline stores evidence in the repo, it publishes that PR-viewable validation evidence to an orphan evidence branch that shares no history with code branches, so it never enters the crew branch or the default branch. This repo uses that setting, and its own `.no-mistakes/` directory remains local state that stays gitignored and is rejected by CI if tracked; [`configuration.md`](configuration.md) owns the setting. PR-based task merges go through `bin/fm-pr-merge.sh`, which records `pr=` and any available `pr_head=` through `bin/fm-pr-check.sh` before calling the forge CLI. The helper requires a full canonical URL and rejects malformed URLs or repo override flags before recording merge state. -A `https://github.com///pull/` URL requires `gh` and `jq`, is merged only after live reads confirm the pull request is open, not a draft, mergeable, conflict-free, every unwaived check is green at the current head, and every unwaived check the base branch requires has reported at that head, then `gh pr merge` binds that verified head with `--match-head-commit`. -A required check that never reported is absent from the checks list rather than red; [`bin/fm-pr-merge.sh`](../bin/fm-pr-merge.sh)'s header owns required-context sources, producer identity, partial-read refusals, and attended check waivers. +A `https://github.com///pull/` URL requires `gh` and `jq`, is merged only after one live read confirms the pull request is open, not a draft, mergeable, conflict-free, and every unwaived check is green at the current head, then `gh pr merge` binds that verified head with `--match-head-commit`. A check run is green when its current run is green, because GitHub leaves a cancelled run in the rollup beside the passing re-run it triggered when the base branch advanced; `bin/fm-pr-merge.sh`'s `github_checks_not_green` owns the rule, which uses `startedAt` to clear only an older completed check run that a passing run with the same name provably replaced, while unfinished check runs and non-green status contexts stay red. `--auto`, `--admin`, and branch-deletion flags are refused unless `--attended-override` is passed for an explicit captain instruction; that override never skips the live green check, the away-record read, or a captain hold. +An attended `--allow-red ` may appear once, waives only GitHub checks with that exact name, and is refused while the away-posture record exists. Because away merge authority is read from that record and then acted on by the forge, the authority read and synchronous forge command share the record's cross-subsystem lock, closing the common live-owner TOCTOU. A lock that cannot be taken refuses the merge. While the record exists, GitHub auto-merge and any base whose rules cannot prove the absence of a merge queue are refused before submission, and GitLab auto-merge flags or scheduled state are refused while an immediate merge is forced with a final `--auto-merge=false`; a branch-rules read that fails only because the repository's plan does not expose branch rules at all (GitHub's plan-upgrade 403) proves the absence of a merge queue on its own and does not refuse, while every other failure to read that state still does. @@ -457,13 +451,16 @@ The [Relay configuration reference](configuration.md#promised-public-replies-sta ## Project memory belongs to projects -Project-memory ownership and the crewmate corrections-only boundary are defined in [`AGENTS.md` section 6](../AGENTS.md#6-project-and-knowledge-management); `data/projects.md` stays a thin private registry. -For manual project initialization, [`bin/fm-ensure-agents-md.sh`](../bin/fm-ensure-agents-md.sh) owns the `CLAUDE.md` pointer, self-governance insertion, and case-variant file refusal; its header and help document the explicit mark for equivalent project-owned guidance. +Durable project-intrinsic agent knowledge lives in each project's committed `AGENTS.md`, with `CLAUDE.md` as a real `@AGENTS.md` import pointer. +Ship briefs prompt crewmates to create or update those files through the normal delivery path; `data/projects.md` stays a thin private registry. +Each project `AGENTS.md` carries self-governance guidance; [`bin/fm-ensure-agents-md.sh`](../bin/fm-ensure-agents-md.sh) owns the canonical wording and idempotent insertion, while its header and help document the explicit mark for equivalent project-owned guidance. +It refuses a case-variant real memory file such as a lowercase `agents.md`, so the pointer's `@AGENTS.md` import resolves to a real `AGENTS.md` on a case-sensitive filesystem, and surfaces the mismatch for manual reconciliation. +The full ownership rule - what is project-intrinsic versus fleet-private, and how firstmate keeps the two apart without writing into project clones - is owned by [`AGENTS.md`](../AGENTS.md) (project and knowledge management). ## Operational memory routing `/stow` sweeps the current session for durable knowledge that only exists in conversation and routes each finding to the most specific disk home. -The destination for each kind of knowledge, including project-intrinsic knowledge, is owned by [`AGENTS.md` section 6](../AGENTS.md#6-project-and-knowledge-management). +Home-domain captain preferences go to `data/captain.md`, cross-domain shared captain preferences go to the primary home's `data/captain-shared.md`, fleet-local operational facts and gotchas go to home-local `data/learnings.md`, project-intrinsic knowledge goes through normal crewmate delivery into that project's committed `AGENTS.md`, and task-scoped notes or undone next steps go to the backlog. Memory writes use inspect-then-update rather than blind append; the internal [`stow` skill](../.agents/skills/stow/SKILL.md) owns tier markers, decay, cold archival, and offload. The same pass also persists open-work record state the session is holding - filing a thread that was never recorded and correcting one the session knows went stale - bounded to the open work that session is actually holding. It is deliberately not a reconciliation of durable records against repository or PR reality: its input is the volatile context, so it can only preserve what the session still knows, and no reconciliation that outlives a session exists today. @@ -495,10 +492,10 @@ The procedure and outcome vocabulary are owned by the [`/updatefirstmate` skill] Fleet state lives in each task's session-provider backend (tmux by hard default, herdr or cmux when selected or auto-detected, zellij/orca when explicitly selected), no-mistakes run records, status event logs, local markdown under `data/` including `data/captain.md`, `data/captain-shared.md`, and `data/learnings.md`, and persistent secondmate homes. For herdr, respawning after a server-restored layout closes and replaces confirmed no-agent or dead task-tab husks instead of requiring manual tab cleanup. -At session start and again on the watcher's bounded liveness cadence, confirmed-dead secondmate agent endpoints are closed and relaunched through the same secondmate spawn path, while ambiguous liveness reads are left untouched to avoid duplicate supervisors. +At session start, confirmed-dead secondmate agent endpoints are closed and relaunched through the same secondmate spawn path, while ambiguous liveness reads are left untouched to avoid duplicate supervisors. Use `/stow` before an intentional reset when the conversation may hold durable knowledge that has not yet been written to disk; after that, the next firstmate session can reconcile and carry on. ## Development notes The current watcher reliability work combines always-on bash triage with a durable queue for actionable wakes, generation-bound post-handling acknowledgement, deterministic re-arm recovery after watcher downtime, a race-proof singleton lock, duplicate self-eviction, drain-time liveness assertion, and a self-verifying tracked-child arm wrapper. -The away posture is the record `bin/fm-afk-contract.sh` owns; see [supervision-host.md](supervision-host.md) for the opt-in non-Pi away session and the `/afk` skill for the remaining daemon-backed harnesses. +The away posture is the record `bin/fm-afk-contract.sh` owns; on the harnesses other than Pi the presence-gated sub-supervisor (`bin/fm-supervise-daemon.sh`) still provides walk-away delivery via the `/afk` skill while reusing the same shared wake classifier as the always-on watcher. diff --git a/docs/configuration.md b/docs/configuration.md index 361becbaf4c..7c300e4f602 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -1,517 +1,180 @@ # Configuration -Configure where Firstmate keeps its files, which tools launch workers, and how supervision runs. -Start with the directory layout, then use the setting reference for the behavior you want to change. +The files and environment variables you set to operate firstmate. -## Find a setting - -| What you want to configure | Start here | -| --- | --- | -| Firstmate's code, private files, or project location | [FM_HOME](#fm_home) and [operational home layout](#operational-home-layout-and-state) | -| Task windows and worker tools | [Runtime backend](#runtime-backend-configbackend--fm_backend) and [harness support](#harness-support) | -| Worker permissions, accounts, or environment | [Claude permission mode](#claude-permission-mode-configclaude-permission-mode), [worker account pin](#worker-account-pin-configclaude-account-configpi-account), and [worker launch environment](#worker-launch-environment-configlaunch-env-allowlist) | -| Backlog, preferences, and memory | [Backlog backend](#backlog-backend-taskstoml--configbacklog-backend), [captain preferences](#captain-preferences-datacaptainmd--datacaptain-sharedmd), and [startup memory budget](#startup-memory-budget-configstartup-memory-budget) | -| Supervision and presentation | [Pi supervision branch](#pi-supervision-branch), [supervision host](#supervision-host-configsupervision-host), and [Calm preference](#calm-preference-configcalm) | -| Persistent secondmates | [Secondmate routes](#secondmate-routes-datasecondmatesmd) | -| Per-run overrides and tuning | [Environment variables](#environment-variables) | - -## FM_HOME - -`FM_HOME` selects the operational home for one firstmate instance. - -| Location | What it contains | Default relationship | -| --- | --- | --- | -| Firstmate repo root | Shared code, including the scripts in this repo's `bin/` | Most scripts also use this as the operational home when `FM_HOME` is unset. | -| Operational home | Private `state/`, `data/`, `config/`, and `projects/` | Selected by `FM_HOME`. | -| Projects directory | Local project clones | Under the operational home; `FM_PROJECTS_OVERRIDE` can select a different directory for tests and specialized harness setup. | - -When `FM_HOME` is unset, most scripts use the repo root as the home. -When it is set, scripts still run from this repo's `bin/`, while `state/`, `data/`, `config/`, and `projects/` come from `$FM_HOME`. - -### Root and directory overrides - -`FM_ROOT_OVERRIDE` overrides the firstmate repo root used by scripts, including the primary checkout watched by the worktree-tangle guard. -When `FM_HOME` is unset, it also behaves as the old whole-root override. - -`bin/fm-send.sh` requires `FM_HOME` to be set before resolving a target. -Unlike most scripts, it does not use the general fallback, because a steer must not silently resolve against the wrong home. -These variables override individual operational directories for tests and specialized harness setup: - -| Variable | Directory selected | -| --- | --- | -| `FM_STATE_OVERRIDE` | Runtime state | -| `FM_DATA_OVERRIDE` | Durable private records | -| `FM_PROJECTS_OVERRIDE` | Local project clones | -| `FM_CONFIG_OVERRIDE` | Local configuration | - -### Relative paths and lifecycle safety - -Before `fm-brief.sh`, `fm-spawn.sh`, or `fm-afk-launch.sh` saves a path or passes it to another process, it handles each applicable `FM_HOME`, `FM_STATE_OVERRIDE`, or `FM_DATA_OVERRIDE` directory as follows: - -- Resolve relative directories against the caller's working directory. -- Preserve accepted absolute spellings unchanged. -- Reject an unresolvable relative directory and name the offending variable. - -`fm-spawn.sh` additionally rejects control bytes in those raw directory inputs before shell or filesystem normalization can change which path the backlog gate checks. -Lifecycle access to a backlog, task record, or pending-close record must resolve within its configured data or state root, and a final-component symlink is refused even when its target remains within that root. - -Bootstrap applies the same relative `FM_HOME` resolution only when embedding that home in the generated Relay poll shim. -Other transient consumers retain their existing shell-relative behavior. - -### Backend labels and containers +## Orchestrator behavior (AGENTS.md) -| Backend | Effect of the operational home | -| --- | --- | -| herdr | `FM_HOME` determines the adapter's workspace label. | -| zellij | `FM_HOME` determines the readable home prefix in visible tab titles, but does not split containers; use `FM_ZELLIJ_SESSION` for a separate session; the full home label also includes a short hash of the resolved `FM_ROOT` path. | -| cmux | `FM_HOME` determines the default config path and readable home prefix in workspace titles; `FM_CONFIG_OVERRIDE` overrides where `config/cmux-socket-password` is read; the full home label also includes a short hash of the resolved `FM_ROOT` path; there is no per-home container split. | +The shared orchestrator behavior lives in [`AGENTS.md`](../AGENTS.md) - edit it like any prompt when the fleet is empty, or dispatch shared-repo edits to a crewmate while tasks are in flight. ## Operational home layout and state -This section is the single owner of the top-level operational-home layout. -Producer script headers and their help own exact child-file fields and mutation contracts. -The tracked code root contains shared instructions, skills, documentation, workflows, and `bin/`. -Each effective `FM_HOME` contains private operational directories. - -`data/` holds durable private fleet records: - -- Project and secondmate registries. -- Captain preferences and optional shared captain preferences. -- Learnings, backlog, briefs, and scout reports. -- Explicitly installed content-addressed extension packages under `data/extensions/packages/`. - -`state/` holds runtime records: - -- Task metadata, append-only status events, and endpoint signals. -- Watcher and wake-queue coordination, away-mode state, and generated Relay artifacts. -- Inactive terminal-outcome receipts under `state/terminal-outcomes/`. -- Enabled extension working namespaces under `state/extensions/`. -- Parent-side remote ledger copies under `state/secondmate-summary-cache/`. -- One-shot Bearings reconcile requests under `state/reconcile-notify/`. -- Private secondmate config-reread generations with their retry and quarantine state. -- Per-task steering-inbox records under `state/.inbox/` (`bin/fm-task-inbox-lib.sh`). -- Parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). - -`config/` holds local gitignored operating choices, including explicit extension bindings under `config/extensions.d/`. - -`projects/` holds local project clones. -Firstmate reads these clones, but changes them only through the narrow guarded and concrete captain-approved exceptions in `AGENTS.md`. +This section is the single owner of the top-level operational-home layout; producer script headers and their help own exact child-file fields and mutation contracts. +The tracked code root contains the shared instruction, skill, documentation, workflow, and `bin/` surfaces, while each effective `FM_HOME` contains private operational directories. +`data/` holds durable private fleet records such as the project and secondmate registries, captain preferences, optional shared captain preferences, learnings, backlog, briefs, scout reports, and explicitly installed content-addressed extension packages under `data/extensions/packages/`. +`state/` holds runtime records such as task metadata, append-only status events, endpoint signals, watcher and wake-queue coordination, inactive terminal-outcome receipts under `state/terminal-outcomes/`, enabled extension working namespaces under `state/extensions/`, away-mode state, generated Relay artifacts, parent-side remote ledger copies under `state/secondmate-summary-cache/`, one-shot Bearings reconcile requests under `state/reconcile-notify/`, private secondmate config-reread generations with their retry and quarantine state, per-task steering-inbox records under `state/.inbox/` (`bin/fm-task-inbox-lib.sh`), and parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). +`config/` holds local gitignored operating choices, including explicit extension bindings under `config/extensions.d/`, and `projects/` holds the local project clones that Firstmate reads but changes only through the narrow guarded and concrete captain-approved exceptions in `AGENTS.md`. Untracked files and directories whose names begin with `scratchpad` are also gitignored, so temporary scratch does not make porcelain-based secondmate sync guards treat a home as dirty. -### Format and lifecycle references - -- `bin/fm-spawn.sh` owns the base task-metadata fields it emits, while the runtime-backend section below owns backend-specific fields and selector interpretation. - -- `bin/fm-contributions.sh` owns durable published-contribution records under each task, observation bounds, equivalent triage-label configuration, and the authenticated contribution check. - -- The producing PR and Relay helpers own the fields they append, [`bin/fm-classify-lib.sh`](../bin/fm-classify-lib.sh) owns status-event vocabulary, optional emission-time syntax, and legacy unknown-time handling, and `bin/fm-crew-state.sh` owns current-state reconciliation. - -- The [`bin/fm-fleet-snapshot.sh` header](../bin/fm-fleet-snapshot.sh) owns the snapshot's event-time and age fields, including secondmate parent-event projections. +`bin/fm-spawn.sh` owns the base task-metadata fields it emits, while the runtime-backend section below owns backend-specific fields and selector interpretation. +`bin/fm-contributions.sh` owns durable published-contribution records under each task, observation bounds, equivalent triage-label configuration, and the authenticated contribution check. +The producing PR and Relay helpers own the fields they append, [`bin/fm-classify-lib.sh`](../bin/fm-classify-lib.sh) owns status-event vocabulary, optional emission-time syntax, and legacy unknown-time handling, and `bin/fm-crew-state.sh` owns current-state reconciliation. +The [`bin/fm-fleet-snapshot.sh` header](../bin/fm-fleet-snapshot.sh) owns the snapshot's event-time and age fields, including secondmate parent-event projections. +Wake, watcher, away-mode, and Relay-specific state mechanics remain with their named scripts and reference sections rather than being duplicated into one exhaustive state tree here. -- Wake, watcher, away-mode, and Relay-specific state mechanics remain with their named scripts and reference sections rather than being duplicated into one exhaustive state tree here. - -### Session-start references - -- `bin/fm-session-start.sh`'s header is the single owner of session-start ordering, composed commands, digest contents, and the digest's startup mechanism. - -- `bin/fm-startup-network.sh`'s header owns the deferred startup stage that keeps every external-network call and the potentially slow inactive-outcome scan off that digest's blocking path, including its state files and the safety argument for running them later. - -- `docs/sessionstart-nudge.md` owns the native session-open adapter tiers that run or nudge the digest command, and the source routing between them. - -- `AGENTS.md` retains the run-once and read-once operator rules, lock-refusal safety, installation consent, and direct-report recovery boundaries because those facts apply at every session start. - -- Ordinary dead-direct-report recovery is owned by `stuck-crewmate-recovery`, while persistent-secondmate recovery is owned by `secondmate-provisioning`. - -## Orchestrator behavior (AGENTS.md) - -The shared orchestrator behavior lives in [`AGENTS.md`](../AGENTS.md). -Edit it like any prompt when the fleet is empty. -While tasks are in flight, dispatch shared-repo edits to a crewmate. +`bin/fm-session-start.sh`'s header is the single owner of session-start ordering, composed commands, digest contents, and the digest's startup mechanism. +`bin/fm-startup-network.sh`'s header owns the deferred startup stage that keeps every external-network call and the potentially slow inactive-outcome scan off that digest's blocking path, including its state files and the safety argument for running them later. +`docs/sessionstart-nudge.md` owns the native session-open adapter tiers that run or nudge the digest command, and the source routing between them. +`AGENTS.md` retains the run-once and read-once operator rules, lock-refusal safety, installation consent, and direct-report recovery boundaries because those facts apply at every session start. +Ordinary dead-direct-report recovery is owned by `stuck-crewmate-recovery`, while persistent-secondmate recovery is owned by `secondmate-provisioning`. ## Calm preference (config/calm) -The Pi Calm extension and the Claude Code Calm mod share the local, gitignored `config/calm` preference under the effective Firstmate home. -One `/calm` choice therefore applies on either harness. -Both resolve the home in this order: `FM_HOME`, `FM_ROOT_OVERRIDE`, then the tracked code root derived from their own path under it. -When `FM_CONFIG_OVERRIDE` is present for tests or specialized setup, it selects the config directory directly. - -### Values and default - -| Value or file state | Result | -| --- | --- | -| `on` | Calm on. | -| `off` | Calm off. | -| Absent, unreadable, or unrecognized | Defaults to off. | - -Both written values end with one newline. -`max` is a legacy value from a removed third presentation level. -Its behavior is now ordinary Calm, so it is still read as `on`. -A home upgraded from `max` keeps Calm on rather than dropping to off. - -### Saving and reloading the preference - -Each `/calm` command saves the new choice before changing live presentation. -A failed write leaves the current choice unchanged and is not reported as a saved preference. -Pi replaces the file atomically; the Claude Code mod uses the plugin API's plain file write. +The Pi Calm extension and the Claude Code Calm mod share the captain's home-local presentation choice in gitignored `config/calm` under the effective Firstmate home, so one `/calm` choice applies on either harness. +Both resolve that home from `FM_HOME`, then `FM_ROOT_OVERRIDE`, then the tracked code root derived from their own path under it, or use `FM_CONFIG_OVERRIDE` as the config directory outright when that test and specialized-setup override is present. +The values they write are `on` and `off`, each followed by one newline; an absent, unreadable, or unrecognized value defaults to off. +`max` is the legacy value written by a removed third presentation level whose behavior is now ordinary Calm, and it is still read as `on`, so a home upgraded from it keeps Calm on rather than dropping to off. +Each `/calm` command persists the new choice before changing live presentation, so a failed write leaves the current choice unchanged rather than claiming persistence; Pi replaces the file atomically, while the Claude Code mod writes it through the plugin API's plain file write. The Pi extension reloads this preference on every Pi `session_start`, including startup, new, resume, fork, and reload reasons. - -The Claude Code mod reloads it on every `session.start`, including same-process session replacement. -It also loads the preference lazily before any row that can draw ahead of that event, including during `claude --continue` restoration. +The Claude Code mod likewise reloads it on every `session.start`, including same-process session replacement, and also loads it lazily before any row that can draw ahead of that event, including during `claude --continue` restoration. This preference is local to each Firstmate home and is not part of secondmate inherited configuration. ## Pi supervision branch -On a Pi primary, an in-process supervision branch handles eligible task-local wake rows and selected heartbeat reviews. -Main-only rows stay on the captain-facing path. -[docs/pi-supervision-branch.md](pi-supervision-branch.md) defines its conversation lifecycle, row eligibility, mixed-queue dispatch, heartbeat routing, and pre-drain recheck. +On a Pi primary, an in-process supervision branch handles eligible task-local wake rows and selected heartbeat reviews while keeping main-only rows on the captain-facing path; [docs/pi-supervision-branch.md](pi-supervision-branch.md) owns its conversation lifecycle, row eligibility, mixed-queue dispatch, heartbeat routing, and pre-drain recheck. Supervision is default-on: once a Pi primary session owns this home's fleet lock, the branch is eligible for every task with no captain grant file required. - -Bash absorbs a genuinely no-op heartbeat before it reaches Pi. -Every watcher-failure alarm stays on the captain-facing main path. -If the branch breaks, wakes still fall back to main in both postures. -The legacy `state/.afk` daemon flag has no effect on Pi. - -### Attended and away authority - -While the away-posture record `state/.afk-contract` exists: - -- The branch takes every actionable row. -- No processing turn opens on the parked main. -- Main's standing authority moves to the branch through the guarded scripts, each retaining its own gate. - -[docs/pi-supervision-branch.md](pi-supervision-branch.md#postures) defines that posture. - -While attended, the branch cannot merge a PR, land local work, freshly spawn, or answer a decision. -These are the bounds set by the captain-approved architecture. -Every existing captain gate remains unchanged in either posture. -Homes on other primary harnesses do not load the Pi branch extension; shared per-task lease behavior is owned by `bin/fm-lease-lib.sh`. - +A genuinely no-op heartbeat is absorbed in bash and never reaches Pi, and every watcher-failure alarm stays on the captain-facing main path. +A broken branch still falls back to today's wake-to-main path in both postures, and the legacy `state/.afk` daemon flag means nothing on Pi. +While the away-posture record `state/.afk-contract` exists the branch takes every actionable row, no processing turn opens on the parked main, and main's standing authority relocates to the branch through the guarded scripts, each keeping its own gate; [docs/pi-supervision-branch.md](pi-supervision-branch.md#postures) owns that posture. +While attended the branch's role stays bounded exactly as the captain-approved architecture set it: it cannot merge a PR, land local work, freshly spawn, or answer a decision, and every existing captain gate remains unchanged in either posture. +Homes on any other primary harness never load this feature and are entirely unaffected. `AGENTS.md`'s `state/` inventory routes the branch's runtime files to their format and lifecycle owners. - -### Outcome delivery and acknowledgement - -While attended, a captain-facing branch outcome (verdict `captain`) is saved as one exact visible transcript entry keyed by sequence. -It then opens one processing turn on main for that sequence. -The turn stays open until main acknowledges the sequence through its `fm_branch_processed` tool. -While away, the entry is saved, but processing waits until the away-posture record is archived. +While attended, a captain-facing (verdict `captain`) branch outcome persists as one exact, sequence-keyed visible transcript entry and then opens one sequence-keyed processing turn on main, which stays open until main acknowledges that sequence through its `fm_branch_processed` tool; while away, the entry persists but processing waits until the record is archived. The branch prompt's "Verdict: routine or captain" section owns the distinction between captain-facing, unsolicited routine, and unchanged-review outcomes. - The generated [Pi supervision protocol](supervision-protocols/pi.md) owns main's event ownership, acknowledgement duty, and conversational treatment for merged outcomes, while the persisted entry itself owns captain visibility. A no-change heartbeat outcome explicitly reported with `task=fleet` and `silent=true` is delivered silently with no rendered note, while every other routine outcome still appends a rendered, sailboat-prefixed note. ## Pi supervision branch model and effort (config/supervision-branch-model, config/supervision-branch-effort) -The branch can run on a cheaper model than main because supervision is an easier job than the captain's own conversation. -It can also use a lower reasoning effort because supervision needs less reasoning than that conversation. - -### Choose a model and effort - +Supervision is an easier job than the captain's own conversation, so the branch can run on a cheaper model than main. +It is also an easier job than the captain's own conversation needs reasoning for, so the branch can run at a shallower effort than main as well. The Pi `/supervision-model` command settles both in one flow: it opens a selector over the models that Pi reports with configured credentials and that this home's stored credentials let the isolated supervision branch resolve, plus a first "Follow main" entry, and then a second picker for the branch's reasoning effort. In Pi's terminal TUI, the model step uses Pi's bounded scrolling list with its input and fuzzy filtering primitives, the same list primitive Pi's `/model` picker scrolls: typing filters the entries, "Follow main" stays the first entry whenever it still matches, and a long catalog scrolls inside the dialog instead of running off the terminal. - The non-TUI RPC, JSON, and print modes have no custom-component surface and keep Pi's generic selector without search, where terminal overflow does not apply. The effort list is a handful of levels and stays on Pi's plain selector dialog. - Both picks change the supervision branch alone and never the captain's own conversation model or effort. - -### Saved settings and available models - -The command saves the model pick in gitignored `config/supervision-branch-model` and the effort pick in gitignored `config/supervision-branch-effort`. -Both live under the effective Firstmate home, resolved in this order: `FM_HOME`, `FM_ROOT_OVERRIDE`, then the tracked code root derived from the extension path. -When `FM_CONFIG_OVERRIDE` is present for tests or specialized setup, it selects the config directory directly. -Firstmate keeps no model catalog of its own. -The list is the intersection of what Pi reports when the picker opens and what a fresh isolated branch runtime can run. - +It persists the model pick in gitignored `config/supervision-branch-model` and the effort pick in gitignored `config/supervision-branch-effort`, both under the effective Firstmate home, resolved from `FM_HOME`, then `FM_ROOT_OVERRIDE`, then the tracked code root derived from the extension path, or under `FM_CONFIG_OVERRIDE` when that test and specialized-setup override is present. +Firstmate keeps no model catalog of its own; the list is the intersection of what Pi reports when the picker opens and what a fresh isolated branch runtime can run. A provider that exists only because an extension registered it inside the captain's session, such as pi-devin-auth's `devin`, is offered and can be pinned or followed like any other; [pi-supervision-branch.md](pi-supervision-branch.md#cost-model-and-the-byte-stable-prefix) owns how that registration reaches the isolated branch runtime. Stored OAuth and API-key credentials retain their native credential type because Firstmate never copies, converts, installs, or overwrites credentials for the branch runtime. - -### Model file format and default - -The model file holds one `/` line followed by one newline. -Parsing splits at the first `/`, so a provider-qualified model id such as `openrouter/anthropic/claude-sonnet-4-5` survives intact. -An absent, unreadable, or unparseable file means no pin. -The branch then follows main's current model, applied explicitly and live whenever main changes models mid-session. - -### Following a native Codex model - +The file holds one `/` line followed by one newline, split at the first `/` so a provider-qualified model id such as `openrouter/anthropic/claude-sonnet-4-5` survives intact. +An absent, unreadable, or unparseable file means no pin, and the branch then follows main's own current model, applied explicitly and live whenever main changes models mid-session. When main uses `codex-native`, following main explicitly selects the same model through ordinary Pi's `openai-codex` provider, so the background branch owns an independent Pi conversation. If that ordinary Pi model is unavailable, the branch refuses to build and returns the notification to main; it never inherits the main native thread or silently selects a different model. - Picking "Follow main" under a `codex-native` main reports that same `openai-codex` model, or that same refusal, because the command and the branch build share one follow rule. A `codex-native` branch pin is refused and excluded from the picker. - -### Applying and changing a model pin - A valid pin wins over main and remains unaffected by main's model changes. Picking "Follow main" removes the file, and the command writes a pin at mode `0600` and replaces it atomically so a failed write leaves the current choice unchanged rather than claiming persistence. - -The file decides the branch model on every build: the new conversation opened at each main session start, and a reopen after a model or effort change within a session. -It overrides the model Pi would otherwise restore from the reopened branch session, so the choice survives both cases. +The file's current state decides the branch model on every branch build - the new conversation each main session start opens and the reopen after a model or effort change inside one session - and it overrides Pi's restore of whatever model a reopened branch session recorded, so the choice survives all of them. That override is what keeps "Follow main" honest: a branch conversation that ran under an earlier pin still records that model, so clearing the file explicitly applies main's model rather than letting the reopened session restore the old one. - -### Unavailable models - -For ordinary Pi providers, an unpinned build passes no model override only when main's model is unknown or this home's stored credentials cannot run it in the isolated branch runtime. -This preserves the behavior from before the file existed. -Model choice never loses the wake. -If main's model could not be applied, the command reports that failure instead of reporting a change that did not take effect. -If a pin names a model Pi cannot return because it is unknown or has no configured credentials, the branch refuses to build. -It sends the accepted wake back to the watcher's captain-facing main path, as any other unreachable branch does. -It never silently falls back to main's model. - +For ordinary Pi providers, only when main's own model is unknown, or this home's stored credentials cannot run it in the isolated branch runtime, does an unpinned build fall back to passing no override at all, which is the behavior from before this file existed; the wake is never lost over model choice, and the command says plainly when main's model could not be applied instead of reporting a change that did not take effect. +A pin naming a model Pi cannot hand back, because the model is unknown or has no configured credentials, is never silently downgraded onto main's model: the branch refuses to build and rejects the accepted wake to the watcher's captain-facing main path, exactly as any other unreachable branch does. Picking also releases the live branch so the next wake reopens this session's own branch conversation under the new model without waiting for a session replacement. -### Effort file format and available levels - -The effort file holds one Pi thinking level followed by one newline. -Model and effort pins are independent: a captain may pin either, both, or neither. -The effort step follows the model step because the effective branch model determines which levels exist. -The menu uses Pi's own supported-level list. -Models without extended levels do not offer them; a non-reasoning model offers only `off`. - +The effort file holds one Pi thinking level followed by one newline, and the two pins are independent: a captain may pin a model, an effort, both, or neither. +The effort step runs after the model step because the effective branch model decides which levels exist: its menu is Pi's own supported-level list, so a model that maps no extended levels simply does not offer them and a non-reasoning model offers only `off`. The picker keeps no effort catalog of its own; when main's model cannot be resolved, it first resolves the model recorded by the most recent branch conversation and uses Pi's supported levels for that effective model. If neither model can be resolved, the picker invents no levels and the command says that the branch's effective effort cannot be determined. - -### Applying and changing an effort pin - -An absent, unreadable, or unrecognized file means no effort pin. -The branch then follows main's current effort, applied explicitly and live whenever main changes effort mid-session. +An absent, unreadable, or unrecognized file means no effort pin, and the branch then follows main's own current effort, applied explicitly and live whenever main changes effort mid-session. A valid pin wins over main and remains unaffected by main's effort changes. - Picking "Follow main" removes the file, and the command writes an effort pin at mode `0600` and replaces it atomically, exactly as it writes a model pin. The effort file's current state decides the branch effort on every branch build, on the same create-and-reopen contract as the model pin and for the same reason: a reopened branch conversation records the effort it last ran under, so only an explicit override keeps "Follow main" honest. - Only when main's own effort cannot be read either does an unpinned build fall back to passing no effort override at all, which is the behavior from before this file existed. - -### Unsupported effort levels - -Pi maps an unsupported pinned level to the branch model's nearest supported level. -Effort never causes the branch to refuse a build. -The captain's raw pick is kept so it applies again on a model that supports it, and the command reports the level the branch will actually use. +Pi owns the clamp, so a pinned level the branch's model cannot run becomes that model's nearest supported level rather than a refusal; the branch is never refused over effort, the captain's raw pick is kept so it applies again on a model that supports it, and the command reports the level the branch will really run at rather than the raw pin. An effort token Pi would not recognize at all is treated as no pin rather than passed to that clamp, which would otherwise collapse a typo into the model's lowest level. -### Cancellation and inheritance - Cancelling the model picker cancels the whole command and changes neither choice. -Cancelling only the effort picker keeps the standing effort choice and still applies the model pick from the same run. -The command's closing message reports both choices as they will take effect. - +Cancelling only the effort picker keeps the standing effort choice and still applies the model pick made in the same run, and the command's one closing message reports both choices as they will actually take effect. Both choices are local to each Firstmate home and are not part of secondmate inherited configuration, the same as the Calm preference; a secondmate home pins its own supervision model and effort with its own `/supervision-model`. -## Supervision host (config/supervision-host) - -The optional local, gitignored `config/supervision-host` enables a supervision host for this home. -The host runs the supervision branch's contract on a headless engine session beside a non-Pi primary. -[docs/supervision-host.md](supervision-host.md) defines its design, current scope, and verified engines. -A Claude, Cursor, OpenCode, omp, Grok, or Codex primary can run the host, only while away. -With the file present, the primary's arm owner runs the host in place of the watcher arm. -The host handles wakes on the engine while `state/.afk-contract` exists. -On that home, `/afk` launches no away daemon; `/quiet` still does. - -Absence leaves the home exactly as it is without the host, on every harness; a Pi primary keeps its in-process supervision branch whether or not the file exists. -A Grok primary reads the file when its session-start block renders, so a change takes effect at its next session start; every other owner reads it at every arm. - -### Engine selection - -The file may be empty, or hold one line ` []`: - -- empty or `default` selects the primary harness's own engine at that engine's default model (`sonnet` for the Claude engine); -- ` []` names a verified engine, currently only `claude`, and optionally the engine's own model name or alias; `default ` selects the primary harness's engine with that model. - -Only Claude has a verified engine of its own, so a Cursor, OpenCode, omp, Grok, or Codex home names `claude` in the file. - -### Failures and when changes apply - -An unverified engine, a primary without a verified engine, or a malformed line leaves the host without an engine. -It takes no wake, so every wake reaches main as it would without the host. -Each away-posture wake includes a line naming the problem. -The file is read at every wake, so a change applies at the next one without a restart. - -It is local to each home and not part of secondmate inherited configuration. -While the file exists, main's lease-checked commands also take the per-task lease lock, so a claim by the host's engine cannot race a mutation main already started (`bin/fm-lease-lib.sh`). - ## Backlog backend (.tasks.toml / config/backlog-backend) The tracked `.tasks.toml` pins the default `tasks-axi` markdown backend to `data/backlog.md`, with `done_keep = 10` and an archive at `data/done-archive.md`. A home may instead select another tasks-axi adapter such as Beads through its own `.tasks.toml` or `TASKS_AXI_BACKEND`; firstmate still uses only tasks-axi verbs for routine backlog reads and mutations, and the adapter maps `start` and evidence-bearing `done` transitions to its native statuses and evidence fields. - -### Captain holds on Beads - Captain-hold row creation is owned by [`bin/fm-captain-hold.sh`](../bin/fm-captain-hold.sh) `hold`: when no work item exists, it creates an ordinary backlog row (`--kind captain` metadata; Beads native type `task`) and then applies the captain hold. Captain rows have no Beads due semantics, so that create path waives a Beads `due.required` setting rather than passing a synthetic `--due`; `--until` remains the optional hold deferral. - Do not register a Beads `types.custom` `captain` type for this: captain is a hold kind, and the fleet Beads `due.required` policy for ordinary work stays in the federated beads config. - -### Automatic dispatch and completion - -When the automatic transition gate applies, dispatch and completion each move the work item in the same run that creates or removes its task record. -The ordinary successful path therefore keeps the backlog and live task set in sync ([`bin/fm-backlog-transition-lib.sh`](../bin/fm-backlog-transition-lib.sh)). +When the automatic transition gate applies, dispatch and completion are not separate operator actions: each moves its work item inside the same run that creates or removes the task's record, so the ordinary successful path cannot leave the backlog and live task set out of sync ([`bin/fm-backlog-transition-lib.sh`](../bin/fm-backlog-transition-lib.sh)). Under that gate, dispatch accepts only an unheld, unblocked Queued or In flight item in this home; a missing, Done, held, or dependency-blocked item is refused before any endpoint or local copy is created. - -[`bin/fm-tasks-axi.sh`](../bin/fm-tasks-axi.sh) refuses `add --start` and its `create --start` alias. -Either would place a row In flight without a task record, status file, or inbox, counting it as live work that nobody is doing. -The wrapper still passes through the documented direct transition `tasks-axi start `. Completion refuses to report success until the item is closed, and session start reconciles this home's own books after an interrupted run. - When a spawn is interrupted after launch delivery began, its exit path re-reads the paired task record and the backlog row under the same per-task lock as the commit, repairs a row the commit believed it had moved, and reports only what was verified or honestly attempted, never intent phrased as outcome ([`bin/fm-spawn.sh`](../bin/fm-spawn.sh); [`tests/fm-backlog-atomicity.test.sh`](../tests/fm-backlog-atomicity.test.sh)). - -### Which backlog receives a transition - Automatic transitions run from the configured data directory's parent, letting that home's effective tasks-axi configuration address its selected adapter while keeping relative scout-report links rooted there. A markdown backlog is additionally addressed by an explicit `--file` at `/backlog.md`, so the change lands in the home that owns the task regardless of the caller's working directory. - Any other configured adapter is addressed by that root alone, because `--file` would override the adapter's own workspace path. - -### Exemptions and refusal conditions - -The gate does not apply to persistent secondmates, manual-backend homes, or markdown homes without a backlog file. -Those retain their existing persistent-agent, manual, or ad-hoc lifecycle behavior. -Configured non-markdown adapters remain active without that file. +The gate does not apply to persistent secondmates, manual-backend homes, or markdown homes without a backlog file, preserving their existing persistent-agent, manual, or ad-hoc lifecycle behavior while configured non-markdown adapters remain active without that file. Migrated-hold resolution on a beads home reads its graph path, binary, and prefix from the root `.tasks.toml` `[beads]` section only, and refuses (rc=2) when the beads backend is selected elsewhere (a `TASKS_AXI_BACKEND` override or user-level config) with no root-level `[beads]` section. - On an automatic-backend home, missing or incompatible `tasks-axi`, an unresolvable configured data directory, or one containing a control byte fails lifecycle work before mutation. An unreadable backend configuration can refuse lifecycle work before the no-backlog exemption applies; repair the configuration named in the diagnostic ([backend resolution contract](../bin/fm-tasks-axi-lib.sh)). - -### Handoffs between homes - Secondmate handoffs bypass that routine-backend choice: `fm-backlog-handoff.sh` keeps only its own fleet-level validation and delegates the item move to `tasks-axi mv`; its [script header](../bin/fm-backlog-handoff.sh) owns route-specific wake outcomes and remote outbox release. It moves in-scope `## Queued` items only and refuses `## In flight` and historical `## Done` records, which stay with their home for pruning or archiving. - Handoff item bodies must use at least two leading spaces, and the helper refuses a selected item with a single-space or tab-indented continuation rather than risk orphaning it. Because bootstrap requires `tasks-axi` on `PATH` on every profile, that delegation works fleet-wide, and the `config/backlog-backend=manual` knob governs firstmate's own hand-editing of its backlog, not this validated helper. - -### Required tools and manual mode - Compatible means the installed build passes the shared version and feature probe owned by [`bin/fm-tasks-axi-lib.sh`](../bin/fm-tasks-axi-lib.sh), including the atomic multi-ID move required by handoff delegation. Bootstrap requires compatible `tasks-axi` on every profile; see "Toolchain" below for missing-tool reporting and silent default-backend behavior. - Set the local, gitignored `config/backlog-backend` file to `manual` to force manual backlog editing and suppress the verbose `BOOTSTRAP_INFO: tasks-axi available` fact, not missing-tool reporting. A `manual` home owns its backlog file outright: the lifecycle transitions above are skipped there, dispatch and completion never fail over the file's contents, and a completed teardown prints the hand edit that is owed instead. - Absent or `tasks-axi` selects the tasks-axi path. On the default markdown adapter, tasks-axi and manual edits produce the same `## In flight`, `## Queued`, and `## Done` sections. -### Using a separate operational home - The tracked `.tasks.toml` paths resolve against the directory tasks-axi runs in, not `FM_HOME`, so a bare `tasks-axi` run from the code root addresses the code root's `data/` whenever the home lives elsewhere. -tasks-axi replaces its target by renaming a temporary file over it. -If the target is a symlink, the write replaces it with a regular file. -Linking the code-root copy into the home therefore forks the queue on the first write instead of keeping the copies in sync. - -Run every routine Firstmate backlog command through [`bin/fm-tasks-axi.sh`](../bin/fm-tasks-axi.sh). -Like lifecycle transitions, it addresses this home's backlog and archive from any working directory. -Bootstrap reports a code-root `data/backlog.md` or `data/done-archive.md` that is not this home's own file as a `BACKLOG_RECONCILE: code-root ...` line, even in a read-only session. +tasks-axi writes by renaming a temp file over its target, which replaces a symlink with a regular file, so linking the code-root copy into the home forks the queue on the first such write rather than keeping the two in step. +Every routine firstmate backlog command therefore runs through [`bin/fm-tasks-axi.sh`](../bin/fm-tasks-axi.sh), which addresses this home's backlog and archive from any working directory exactly as lifecycle transitions do, and bootstrap reports a code-root `data/backlog.md` or `data/done-archive.md` that is not this home's own file as a `BACKLOG_RECONCILE: code-root ...` line even in a read-only session. ## Runtime backend (config/backend / FM_BACKEND) For spawn-capable adapters, the runtime session-provider backend controls where task windows/endpoints are created, captured, sent to, watched, and killed. - -| Runtime backend | Verification status | Reference | -| --- | --- | --- | -| `tmux` | Verified reference backend | [`docs/tmux-backend.md`](tmux-backend.md) | -| `herdr` | Has its own required CI lane | [`docs/herdr-backend.md`](herdr-backend.md) | -| `zellij` | Experimental; no dedicated real-backend CI lane | [`docs/zellij-backend.md`](zellij-backend.md) | -| `orca` | Experimental; no dedicated real-backend CI lane | [`docs/orca-backend.md`](orca-backend.md) | -| `cmux` | Experimental; no dedicated real-backend CI lane | [`docs/cmux-backend.md`](cmux-backend.md) | - +`tmux` is the verified reference backend (see [`docs/tmux-backend.md`](tmux-backend.md)); `herdr` has its own required CI lane (see [`docs/herdr-backend.md`](herdr-backend.md)); `zellij`, `orca`, and `cmux` remain experimental spawn backends with no dedicated real-backend CI lane (see [`docs/zellij-backend.md`](zellij-backend.md), [`docs/orca-backend.md`](orca-backend.md), and [`docs/cmux-backend.md`](cmux-backend.md)). Treehouse remains the worktree provider for tmux, herdr, zellij, and cmux, since herdr, zellij, and cmux are session providers only; Orca provides both the task worktree and terminal endpoint. - -### Backend selection order - -New spawns choose the backend in this order: - -1. An explicit `--backend` flag authorized for that exact task by a present captain instruction or the task's own accepted brief. - A later task cannot inherit that authority by analogy. -2. `FM_BACKEND`. -3. The first non-empty line of local, gitignored `config/backend`. -4. Runtime auto-detection from `$TMUX`, `HERDR_ENV=1`, or cmux runtime signals. -5. Default `tmux`. - +New spawns choose the backend in this order: an explicit `--backend` flag that current authority for that exact task alone has authorized (a present captain instruction or the task's own accepted brief; never later-task precedent by analogy), then `FM_BACKEND`, then the first non-empty line of local gitignored `config/backend`, then runtime auto-detection from `$TMUX`, `HERDR_ENV=1`, or cmux runtime signals, then default `tmux`. If more than one runtime marker is present, detection resolves innermost-first: `$TMUX` is checked before `HERDR_ENV=1`, which is checked before cmux's primary `CMUX_WORKSPACE_ID` marker and its documented fallback signals - tmux or herdr started from inside a cmux terminal is the innermost, currently-executing layer, while cmux itself (a terminal application, not a nestable multiplexer) is always checked last. See [`docs/cmux-backend.md`](cmux-backend.md#runtime-detection) for why cmux can be selected when `CMUX_WORKSPACE_ID` is absent. - Auto-detected Herdr stays silent like tmux, while auto-detected cmux prints a stderr notice naming `config/backend` and `--backend tmux` because cmux remains experimental. Zellij and Orca are never auto-detected; select them by putting the name in a local `config/backend` file, by exporting `FM_BACKEND=`, or by telling the first mate in chat. - -### Accepted backends and secondmate limits - Any value other than `tmux`, `herdr`, `zellij`, `orca`, or `cmux` is rejected until another adapter is implemented and verified. `fm-spawn.sh` accepts `tmux`, `herdr`, `zellij`, `orca`, and `cmux` for ship and scout tasks; `backend=orca` and `backend=cmux` both still refuse `--secondmate` until secondmate launch semantics are designed for each. - `codex-app` is not an accepted runtime backend yet; [`docs/codex-app-backend.md`](codex-app-backend.md) owns the Codex App boundary. - -### Liveness classification - -The session-start secondmate liveness sweep and the watcher's secondmate liveness tick use the recovery-grade `fm_backend_agent_state` classifier where verified. +The session-start secondmate liveness sweep uses the recovery-grade `fm_backend_agent_state` classifier where verified. The comment above that function in `bin/fm-backend.sh` is the single owner of its detailed state contract and recovery authorization. - The compatibility helper `fm_backend_agent_alive` continues to collapse those detailed results to `alive`, `dead`, or `unknown` for older callers. - -### Dependency and socket checks - -- A herdr spawn additionally version-gates against the installed `herdr` binary's protocol and requires `jq`, refusing loudly on an incompatible or missing installation. - -- A zellij spawn additionally version-gates against the installed `zellij` binary's version and requires `jq`, refusing loudly when either is missing or the version is older than 0.44. - -- A cmux spawn additionally version-gates against the installed `cmux` binary's version, requires `jq`, and requires the control socket to be reachable and accessible (see [`docs/cmux-backend.md`](cmux-backend.md) "Setup" for the one-time socket-access configuration this needs; Automation mode is the recommended socket control mode, with Password mode supported via `config/cmux-socket-password`), refusing loudly and non-retryably on a `cmuxOnly`/unauthenticated socket. - +A herdr spawn additionally version-gates against the installed `herdr` binary's protocol and requires `jq`, refusing loudly on an incompatible or missing installation. +A zellij spawn additionally version-gates against the installed `zellij` binary's version and requires `jq`, refusing loudly when either is missing or the version is older than 0.44. +A cmux spawn additionally version-gates against the installed `cmux` binary's version, requires `jq`, and requires the control socket to be reachable and accessible (see [`docs/cmux-backend.md`](cmux-backend.md) "Setup" for the one-time socket-access configuration this needs; Automation mode is the recommended socket control mode, with Password mode supported via `config/cmux-socket-password`), refusing loudly and non-retryably on a `cmuxOnly`/unauthenticated socket. A backend spawn refusal from a missing dependency, version gate, or unauthenticated socket is terminal for that selected backend; firstmate surfaces it as a blocker instead of silently retrying another backend. - -### Task metadata - Task meta records `backend=` only for a non-default backend; an absent `backend=` means `tmux`, preserving existing default-path meta files. - -- Every new task records `endpoint_task_id=` as the cleanup binding between the metadata filename and its opaque runtime endpoint. - -- A herdr task additionally records `herdr_session=`, `herdr_workspace_id=`, `herdr_tab_id=`, and `herdr_pane_id=`. - -- A zellij task additionally records `zellij_session=`, `zellij_tab_id=`, and `zellij_pane_id=`. -- An Orca task additionally records `orca_worktree_id=` and `terminal=`, with `window=fm-` kept as the shared firstmate alias. - -- A cmux task additionally records `cmux_workspace_id=` and `cmux_surface_id=`. - -### Task selectors - +Every new task records `endpoint_task_id=` as the cleanup binding between the metadata filename and its opaque runtime endpoint. +A herdr task additionally records `herdr_session=`, `herdr_workspace_id=`, `herdr_tab_id=`, and `herdr_pane_id=`. +A zellij task additionally records `zellij_session=`, `zellij_tab_id=`, and `zellij_pane_id=`. +An Orca task additionally records `orca_worktree_id=` and `terminal=`, with `window=fm-` kept as the shared firstmate alias. +A cmux task additionally records `cmux_workspace_id=` and `cmux_surface_id=`. Task selectors for `fm-peek.sh`, `fm-send.sh`, and `fm-crew-state.sh` resolve centrally through `fm_backend_resolve_selector`. A selector containing `:` is passed through as an explicit backend endpoint escape hatch. - Otherwise an exact task id matching `state/.meta` wins before the legacy `fm-` label fallback, so task ids that themselves start with `fm-` route to their own metadata instead of being stripped. A metadata-routed selector returns the recorded backend target (`terminal=` for Orca, otherwise `window=`), and matching explicit targets can still recover the recorded backend when metadata contains the same endpoint. - Only metadata-routed task selectors carry secondmate-marker and Codex-harness context; explicit endpoint escape hatches do not. -These rules are the single owner of the task-selector vocabulary. -Backend guides and other documents refer here instead of restating the resolution order. - -### Teardown identity checks - +These five sentences are the single owner of the task-selector vocabulary; backend guides and other documents point here instead of restating the resolution order. `fm-teardown.sh ` takes a task id directly and validates the complete metadata-only endpoint identity before any runtime dispatch or cleanup mutation. Missing, empty, duplicate, malformed, backend-inconsistent, or task-mismatched endpoint records are preserved and refused. - Legacy tmux metadata remains cleanup-compatible when its exact window name is `fm-`; opaque non-tmux endpoints require their recorded `endpoint_task_id=` binding. - -### Herdr homes and presentation - `FM_HOME` determines Herdr's home label: the primary home uses `firstmate`, and a secondmate home marked by `.fm-secondmate-home` uses `2ndmate-`. [`herdr-backend.md`](herdr-backend.md#watching-and-task-containers) owns launcher-bound workspace placement, the label-only fallback, collision handling, and recovery behavior. - The local `config/herdr-presentation-spaces` file instead opts a home out of, or explicitly in to, Herdr's default-on disposable single-task visual projection; [Presentation spaces](herdr-backend.md#presentation-spaces) owns its accepted values, default, Herdr version floor, migration, behavior, safety limits, recovery contract, and narrow locked session-start cleanup of exact restored idle-shell children. The setting is inherited into secondmate homes under the primary-authoritative contract owned by [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). - For normal herdr operations, `HERDR_SESSION` selects the named session, but destructive test cleanup must not rely on `HERDR_SESSION` alone. Use the explicit guarded cleanup path described in [`docs/herdr-backend.md`](herdr-backend.md) instead of `herdr server stop`. - -### Zellij sessions - For normal zellij operations, `FM_ZELLIJ_SESSION` selects the named session and defaults to `firstmate`. Zellij has no per-home workspace split: primary and secondmate tasks share that one session, and visible tab titles are scoped by the active `FM_HOME` readable label plus a short hash of the resolved `FM_ROOT` path as `fm--`. - Use the guarded cleanup path described in [`docs/zellij-backend.md`](zellij-backend.md) instead of `kill-all-sessions` or `delete-all-sessions`. - -### cmux workspaces - cmux has no session layer at all - one workspace per task, in whatever cmux window is open - and its socket password (when configured) is read from local, gitignored `config/cmux-socket-password` under the effective config directory, never committed. The caller-facing label remains `fm-`, but the actual cmux workspace title is scoped by the active `FM_HOME` readable label plus a short hash of the resolved `FM_ROOT` path as `fm--`. - Test cleanup must use the guarded path in [`docs/cmux-backend.md`](cmux-backend.md#current-operation-and-safety), never enumerate-and-close every workspace. `config/backend` is inherited into secondmate homes under the primary-authoritative contract owned by [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). @@ -519,45 +182,30 @@ Test cleanup must use the guarded path in [`docs/cmux-backend.md`](cmux-backend. The `/afk` sub-supervisor injects escalation digests into firstmate's own pane independently of where new task endpoints are spawned. It currently supports only `tmux` and `herdr` supervisor panes. - Set `FM_SUPERVISOR_BACKEND=tmux|herdr` and `FM_SUPERVISOR_TARGET=` to override both axes explicitly; for herdr the target is `":"`. Without overrides, backend detection uses `$TMUX_PANE` first, then `HERDR_ENV=1` with `HERDR_PANE_ID`, then falls back to `tmux`. - That keeps a tmux pane nested inside herdr on the tmux transport, matching the runtime backend's innermost-first rule. Target detection uses `FM_SUPERVISOR_TARGET`, then `$TMUX_PANE`, then `"${HERDR_SESSION:-default}:${HERDR_PANE_ID}"` under herdr, then the legacy `firstmate:0` tmux fallback with a warning. - Selecting any other supervisor backend, including `zellij`, `orca`, or `cmux`, refuses at daemon startup instead of trying tmux injection primitives against a non-tmux pane. ## Away-mode wedge alarm channels (config/wedge-alarm) When away-mode injection wedges past `FM_MAX_DEFER_SECS`, the sub-supervisor raises a loud, rate-limited alarm. Beyond the durable `state/.subsuper-inject-wedged` marker and the tmux status-line flash, it attempts a configured backend-independent active alert that can reach the captain even when every pane and its backend status-line is unreadable. - -### Channels and overrides - `config/wedge-alarm` (local, gitignored) lists channel directives, one per non-empty, non-comment line; every listed non-`off` channel fires, best-effort. `FM_WEDGE_ALARM_CHANNEL` overrides the file with a single directive. - Directives are `off` (a position-independent kill switch that disables every active alert), `auto`/`default`, `osascript` (macOS Notification Center banner), `herdr` (herdr UI notification), and `command:` (run `` via `sh -c`, summary on `$1` and stdin). An absent file means `auto`, i.e. default-on on macOS: the alarm exists precisely so a wedged away-mode primary is never silent, and it fires at most once per max-defer window after a genuine wedge. - A missing or failing channel logs and falls through to the next, never crashing the daemon. See [`wedge-alarm.md`](wedge-alarm.md) for the current channel reference, [`verification/supervision.md`](verification/supervision.md#wedge-alarm-channels) for active evidence, and [`examples/wedge-alarm`](examples/wedge-alarm) for a copyable config. ## Trace context propagation (config/trace-context / FM_TRACE_CONTEXT) The optional local, gitignored `config/trace-context` presence flag enables default-off native W3C trace-context propagation. - -### Precedence and session boundary - `FM_TRACE_CONTEXT` overrides the file: `1`/`on`/`true`/`yes` enables, any other non-empty value disables, and unset or empty defers to the file. Each locked home session resolves those inputs once, and all spawns from that home use the frozen decision until a new session starts. - -### Secondmate propagation - When launching a Secondmate, the primary copies the presence flag into its home and passes the primary session's frozen decision as a non-empty `FM_TRACE_CONTEXT=on|off` override for the Secondmate's own session start. A Secondmate on a remote route is covered the same way: the primary resolves and records that task's carrier, and the configured host exports it and receives the same enablement snapshot. - The presence flag is session-scoped enablement, so it transfers at launch and is left unchanged by live convergence into a running home. See [`trace-context.md`](trace-context.md) for carrier semantics, supported routes, the manual fleet-restart requirement, the session boundary, and safety limits; `bin/fm-trace-context-lib.sh`'s header owns the exact mechanics, and [`verification/trace-context.md`](verification/trace-context.md) records repeatable evidence. @@ -569,50 +217,36 @@ See [`fleet-ledger.md`](fleet-ledger.md) for the opt-in setup, record contract, The optional local, gitignored `config/turnend-churn-absorb` presence flag opts this home into a default-off third form of positive work evidence in watcher triage. With it present, every referenced task must independently show positive work evidence, and an eligible bare turn-ended task that lacks authoritative proof may satisfy that requirement when its pane content changed since the previous poll. - -### Evidence and time limit - It stays opt-in because the other two proofs read a verdict the harness itself vouches for while this one infers execution from rendered bytes; with the flag absent triage behaves exactly as it did before. `FM_TURNEND_CHURN_ABSORB_SECS` is a positive integer number of seconds, defaults to `900`, and bounds how long one endpoint's turn-ends may ride that evidence before surfacing anyway. - An invalid value fails closed and surfaces the wake. The bound is required rather than cosmetic because churn and pane staleness read the same pane. - The flag is a home-local supervision-noise preference and is not inherited by secondmate homes, which run their own crew mix. [`architecture.md`](architecture.md) owns the triage contract and `bin/fm-watch.sh`'s `signal_turnend_panes_churned` owns the exact evidence and fail-closed boundaries. ## Parked-gate wait deferral (config/wedge-defer-parked-gate) The optional local, gitignored `config/wedge-defer-parked-gate` presence flag opts this home into a default-off second form of wait evidence in the watcher's wedge timer. - -### When a waiting gate defers an alarm - With it present, a provably-working pane about to escalate is also deferred to the `FM_PAUSE_RESURFACE_SECS` recheck cadence when its crew's own current state is a validation gate whose answer is owed to the supervisor and whose decision for that run is still open, and the recheck names the supervisor and the action that clears the lane instead of reporting a suspected wedge. It stays opt-in because the other evidence is the worker's own declaration about its own silence, while this is derived from a pipeline's gate state, so which lanes give up the escalation ladder for it is a home's choice. - With the flag absent the wedge timer spends no fold or current-state read for it, writes no record, and keeps the unchanged escalation schedule, reasons, and `demand-deep-inspection` wording. The flag is a home-local supervision-noise preference and is not inherited by secondmate homes, which supervise their own crew and own that trade separately. - [`architecture.md`](architecture.md) owns the wait-evidence contract and which records may take the ladder away; `bin/fm-watch.sh`'s `wedge_wait_evidence` owns the exact derivation and its fail-closed boundaries. ## Gate defaults (.no-mistakes.yaml) The tracked `.no-mistakes.yaml` sets `test.evidence.store_in_repo: true` and pins `commands.lint` to `bin/fm-lint.sh`, the same owner CI invokes. Storing evidence in the repo publishes each run's test artifacts to the orphan `no-mistakes/evidence` branch and links them from the PR body, instead of keeping them on local disk under the no-mistakes home. - That branch shares no history with code branches, so evidence never enters a pushed feature branch or the default branch; the worktree's `.no-mistakes/` stays local and CI rejects tracked entries under that path. The [`firstmate-coding-guidelines` skill](../.agents/skills/firstmate-coding-guidelines/SKILL.md#no-mistakes-test-configuration) owns why `commands.test` stays absent and targeted validation belongs to the evidence path. - `commands.test` executes code, so no-mistakes honors it only from the default-branch copy of `.no-mistakes.yaml`; a pushed branch cannot change what the gate runs. See [CONTRIBUTING.md](../CONTRIBUTING.md) for the firstmate-specific local test policy and entry points. - Portable shard evidence and coverage rules are in [fm-test-portable-shards.md](fm-test-portable-shards.md); [herdr-backend.md](herdr-backend.md#destructive-lab-safety) owns the real-Herdr lane's isolation boundary, and [runtime-backends.md](verification/runtime-backends.md#herdr) owns active evidence. ## Captain Preferences (data/captain.md / data/captain-shared.md) Domain-local preferences for one captain's fleet live locally in each home's `data/captain.md`; it is gitignored and printed in the session-start context digest after `data/projects.md` and optional `data/secondmates.md`. Before changing it, inspect the current file and curate the matching bullet in place under the internal [`stow` skill's](../.agents/skills/stow/SKILL.md) tiering and archive contract; add a new bullet only for a genuinely new durable preference. - Shared captain preferences that apply across secondmate domains live only in the primary home's optional `data/captain-shared.md`. `secondmate-provisioning` owns its propagation contract, including the required header, read-only secondmate copies, quarantine diagnostics, and the rollout rule that existing homes trim `data/captain.md` by hand after first propagation rather than deleting private content automatically. @@ -620,295 +254,165 @@ Shared captain preferences that apply across secondmate domains live only in the Fleet-local operational facts and gotchas live locally in `data/learnings.md`; it is gitignored and printed after the captain-preference files in the session-start context digest. The file is created lazily on first learning and follows the internal [`stow` skill's](../.agents/skills/stow/SKILL.md) aging-tier and cold-archive contract: inspect the current file first and curate it instead of appending forever. - There is no shared learnings file by captain decision. ## Startup memory budget (config/startup-memory-budget) `config/startup-memory-budget` is the primary-authoritative per-home allowance for the startup prompt-memory surface: `data/captain.md`, `data/captain-shared.md`, and `data/learnings.md` together. The locked mutable bootstrap path materializes its visible default of `7500` estimated tokens in a primary home when the file is absent. - -### Set and validate the budget - To select another allowance, replace the primary home's file with one valid positive value in the exact format below; the next locked bootstrap convergence or `bin/fm-config-push.sh` propagates it to registered secondmates. A secondmate does not create an independent default and instead receives the primary value through the inherited-local-material contract in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). - The file must be one positive base-10 integer followed by exactly one newline in a regular, single-linked file beneath a non-symlinked `config/` directory. Malformed, multi-line, symlinked, hardlinked, special, or otherwise unsafe values are rejected rather than treated as a default. - -### Accounting and curation - Use `bin/fm-startup-memory-budget.sh read` to validate and print the effective value, or `bin/fm-startup-memory-budget.sh report` to account for the three files. The stable local estimate is `ceil(UTF-8 bytes / 3)` per file, a conservative portable approximation rather than a provider-exact tokenizer. - An inherited `data/captain-shared.md` counts in a secondmate's total but remains primary-owned and read-only there. The internal [`/stow` skill](../.agents/skills/stow/SKILL.md) owns curation and its automatic secondmate cascade, which accounts every home against this same per-home allowance separately rather than against a fleet total. - The helper's header owns exact parsing, publication, and report output mechanics. ## Stow pass horizon (config/stow-pass-horizon) `config/stow-pass-horizon` is an optional local, gitignored presence flag that opts this home in to the pass-count decay horizon in the internal [`/stow` skill](../.agents/skills/stow/SKILL.md). Without it a `/stow` pass decays memory entries on their wall-clock horizons alone - 30 days for `aging`, 7 days for `perishable` - which is the default and unchanged behavior. - With it, an entry is also stale after 10 passes (`aging`) or 3 passes (`perishable`) that evaluated it without reinforcing it, whichever horizon it reaches first. Opt in for a home that stows often enough that entries never sit unreinforced for a wall-clock horizon, so memory only grows against the startup-memory budget above; a home that stows rarely already exceeds its date horizon on a single pass and gains nothing. - The flag is per home and is not inherited by secondmate homes, because stow cadence is a property of the home doing the stowing. Only the file's presence is read, so its contents are ignored; remove it to return to the default contract on the next pass. - The skill text owns the marker spelling, the tick order, and the reinforcement rule. ## Secondmate routes (data/secondmates.md) Persistent secondmate routes live locally in `data/secondmates.md`. The concise single-line route contract is owned by the [`secondmate-provisioning` skill](../.agents/skills/secondmate-provisioning/SKILL.md#routing-table), including the parser-compatible fields, one-sentence summary requirement, `home:` pointer to the seeded charter, and limit on extra registry prose. - -### Remote routes and validation - A remote route adds `host:` and `root:` before the existing fields and places the whole secondmate home on that SSH host; it does not make ordinary workers remotely placeable. [`remote-secondmates.md`](remote-secondmates.md) owns current remote setup, operation, and safety behavior. - Use `fm-home-seed.sh validate` to check the complete operational registry contract documented by the command itself. The main first mate routes by reading those scopes with judgment; the project list is provisioning data, not exclusive ownership. - -### Provision a local home - Use `fm-home-seed.sh - {...|--no-projects}` to lease a fresh local firstmate worktree for the secondmate home. For remote provisioning, including supplied project origins, follow [Remote second mates](remote-secondmates.md#provision-a-route). - Use the deliberate `--no-projects` signal only for a firstmate-repo domain that needs no separate project clones. It cannot be combined with a project list, and omitting both still fails loudly. - A project-less seed requires no existing project clones or `data/projects.md` entries in the home, so it refuses a populated-home conversion without changing that home. A preexisting project-bearing charter is also refused until it is re-scaffolded with `--no-projects` or removed. - The lease is held under the secondmate id until explicit retirement or seed rollback returns it, so normal restarts do not free or recycle the home. Teardown of a leased home fails closed if `treehouse return` cannot release the lease; plain-clone homes with no treehouse pool slot are removed directly. - -### Project modes and backlog handoff - Secondmate routes cover `no-mistakes` and `direct-PR` projects; `local-only` projects remain main-firstmate work. For `no-mistakes` projects, seeding initializes only projects newly cloned into a secondmate home and refuses to mutate a preexisting clone that is not already initialized. - After creating a secondmate, move existing main-backlog queued items that you have judged in-scope with `fm-backlog-handoff.sh ...`; it refuses In flight, Done, or non-secondmate homes, and its [script header](../bin/fm-backlog-handoff.sh) owns route-specific wake outcomes and retries. Set `FM_SECONDMATE_CHARTER` to seed from inline charter text when no filled charter brief exists; set `FM_SECONDMATE_SCOPE` when the routing scope should differ from the charter text. - The seeded home's `data/charter.md` owns the standard secondmate lifecycle and escalation contract; the route file points to it through the existing `home:` field instead of adding another pointer. - -### Identity markers and upgrades - Each seed writes an `.fm-secondmate-home` identity marker at the home root, alongside a durable `.fm-secondmate-parent` record of the home's route to its parent (see "Provision a route" in [`docs/remote-secondmates.md`](remote-secondmates.md)). The tracked root `.gitignore` ignores both markers, so validation can read them without making a freshly seeded home appear dirty to porcelain-based safety checks. - This does not relax protection for any other untracked file. An existing linked-worktree home that predates this rule advances through its marker-only state during its next bootstrap or spawn local sync, after which Git ignores the marker normally. - A local standalone-clone home cannot receive a primary-local commit through that no-fetch sync, so it receives the rule through `/updatefirstmate`'s origin refresh instead. -## Harness support +## FM_HOME -claude, codex, opencode, pi, pi-signed, grok, kimi, cursor, and omp are empirically verified for crewmate and secondmate launches; gemini is verified for crewmate and scout launches only, and [README requirements](../README.md#requirements) own the set supported for the primary session. +`FM_HOME` selects the operational home for one firstmate instance. +When it is unset, most scripts use the repo root as the home; when it is set, scripts still run from this repo's `bin/`, but `state/`, `data/`, `config/`, and `projects/` come from `$FM_HOME`. +`FM_ROOT_OVERRIDE` overrides the firstmate repo root used by scripts, including the primary checkout watched by the worktree-tangle guard. +When `FM_HOME` is unset, it also behaves as the old whole-root override. +`bin/fm-send.sh` is intentionally stricter than that general fallback: it requires `FM_HOME` to be set before resolving a target, so operator steers cannot silently resolve against the wrong home. +`FM_STATE_OVERRIDE`, `FM_DATA_OVERRIDE`, `FM_PROJECTS_OVERRIDE`, and `FM_CONFIG_OVERRIDE` override individual operational directories for tests and specialized harness setup. +Before `fm-brief.sh`, `fm-spawn.sh`, or `fm-afk-launch.sh` persists a path or passes it to another process, it resolves each applicable relative `FM_HOME`, `FM_STATE_OVERRIDE`, or `FM_DATA_OVERRIDE` directory against the caller's working directory, preserves accepted absolute spellings unchanged, and rejects an unresolvable relative directory with the offending variable named. +`fm-spawn.sh` additionally rejects control bytes in those raw directory inputs before shell or filesystem normalization can change which path the backlog gate checks. +Lifecycle access to a backlog, task record, or pending-close record must resolve within its configured data or state root, and a final-component symlink is refused even when its target remains within that root. +Bootstrap applies the same relative `FM_HOME` resolution only when embedding that home in the generated Relay poll shim; other transient consumers retain their existing shell-relative behavior. +For the herdr backend, `FM_HOME` also determines the workspace label used by the adapter. +For the zellij backend, `FM_HOME` does not split containers, but it determines the readable home prefix embedded in visible tab titles; use `FM_ZELLIJ_SESSION` when a separate zellij session is needed. +The full zellij home label also includes a short hash of the resolved `FM_ROOT` path. +For the cmux backend, `FM_CONFIG_OVERRIDE` overrides where `config/cmux-socket-password` is read from, while `FM_HOME` determines the default config path and readable home prefix embedded in workspace titles. +The full cmux home label also includes a short hash of the resolved `FM_ROOT` path, and there is no per-home container split. -### Harness restrictions and credentials +## Harness support +claude, codex, opencode, pi, pi-signed, grok, kimi, cursor, and omp are empirically verified for crewmate and secondmate launches; gemini is verified for crewmate and scout launches only, and [README requirements](../README.md#requirements) own the set supported for the primary session. `fm-spawn.sh` refuses kimi on cmux and Orca at preflight, because answering Kimi's folder-trust dialog needs a verified viewport-only capture those backends lack; [its adapter reference](../.agents/skills/harness-adapters/references/harness/kimi.md#readiness-gated-start) owns the trust-dialog handling. A cursor secondmate or primary runs the tracked project-scope `.cursor/hooks.json` in its own home and must be launched with `--trust`, or no project hook loads; [`docs/supervision-protocols/cursor.md`](supervision-protocols/cursor.md) owns its supervision protocol. - Cursor typed-submit confirmation is verified on tmux and Herdr only. On Zellij, cmux, and Orca a typed-plane Cursor send (a harness-native invocation or an explicit backend target; ordinary text steers ride the durable inbox and exit 0 at enqueue) lands, but `fm-send` reports delivery unconfirmed and exits non-zero because their shared submit core does not consult the busy footer; [runtime backend verification](verification/runtime-backends.md#cursor-agent-cli) owns the evidence and transcript-state boundary. - muse is verified for crewmate and scout launches ONLY, and `fm-spawn.sh` refuses it for a secondmate, because muse ships no usable hook surface for a primary session's turn-end supervision; [`docs/verification/muse.md`](verification/muse.md) owns that evidence. muse also needs a worker-reachable credential before spawning, and the portable fleet path is the `/muse/auth.json` credential stored by `muse login`, because a caller-only `META_API_KEY` does not cross a long-lived backend daemon. - gemini is likewise refused for secondmates because it has no primary supervision protocol; [its adapter reference](../.agents/skills/harness-adapters/references/harness/gemini.md) owns the credential precondition, canonical-launch wiring, and raw-launch limitations. rovo is likewise verified for crewmate and scout launches ONLY, refused for a secondmate for the same reason - no turn-end hook and no primary supervision protocol; [`docs/verification/rovo.md`](verification/rovo.md) owns that evidence, including the OAuth token's silent background refresh from a stored refresh token and both tmux and herdr pane liveness (herdr placement is verified live, with a Herdr-side agent-detection gap left open for recovery classification). - agy is likewise verified for crewmate and scout launches ONLY, refused for a secondmate for the same reason - no hook surface and no primary supervision protocol; [`docs/verification/agy.md`](verification/agy.md) owns that evidence, including the spawn-time worktree trust pre-registration through `bin/fm-agy-trust.sh` and Herdr's native agy pane recognition. devin is verified for crewmate and scout launches only; a secondmate is refused because Devin has no verified primary supervision protocol. - Its private worker config disables Claude Code imports (including the captain's hooks) and Devin commit attribution without editing user or project config; [`fm-devin-config.sh`](../bin/fm-devin-config.sh) owns these enforced settings and [Devin verification](verification/devin.md) owns the live evidence and observed model availability. - -### Verification and primary supervision - New harnesses get verified through a supervised trial task before joining the set. The verified adapter evidence - each harness's busy-state source, interrupt and exit behavior, skill-invocation syntax, and per-harness quirks - lives in the skill tree rooted at [`.agents/skills/harness-adapters/SKILL.md`](../.agents/skills/harness-adapters/SKILL.md). - The executable interrupt and exit mechanics live in [`bin/fm-control-lib.sh`](../bin/fm-control-lib.sh), and [`docs/agent-control.md`](agent-control.md) owns their lifecycle-control architecture. Launch mechanics, including the verified command templates, live in [`bin/fm-spawn.sh`](../bin/fm-spawn.sh). - Pi-family launches adapt the regular-TUI safeguard to the installed CLI's capabilities; [`fm-spawn.sh --help`](../bin/fm-spawn.sh) owns the exact version-safe launch mechanics. Enabled primary-session turn-end guard integrations are tracked as repo-level hook files and documented in [`docs/turnend-guard.md`](turnend-guard.md). - Kimi remains outside the primary turn-end guard integrations; [`docs/turnend-guard.md`](turnend-guard.md#compatibility-limits) owns its separate captain-approved crew wake hook. Primary-session watcher wake protocols are rendered at session start by [`bin/fm-supervision-instructions.sh`](../bin/fm-supervision-instructions.sh) from [`docs/supervision-protocols/`](supervision-protocols/). - Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Cursor's stop hook parks on the watcher, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi and pi-signed use the same two tracked primary extensions, omp uses its own two tracked `.omp/extensions/` files with a blocking `session_stop` turn-end hook, and OpenCode uses its TUI plugin. - -### Choose the worker harness - `config/crew-harness` is a local, gitignored file containing one adapter name for crewmate and scout launches. When pi-signed is selected, Firstmate preserves `FM_PI_HARNESS=pi-signed` and refuses the launch if the selected executable is unavailable rather than falling back to pi; [`fm-spawn.sh --help`](../bin/fm-spawn.sh) owns executable resolution and launch mechanics. - Plain Pi launches set `FM_PI_HARNESS=pi`, so a signed primary's environment cannot relabel a plain Pi worker. When it is absent or contains `default`, crewmates mirror the firstmate's own harness. - -### Choose the secondmate harness - `config/secondmate-harness` is a separate local, gitignored file containing the adapter the primary uses to launch secondmate agents, optionally followed by model and effort tokens on the same line. The first non-empty, non-comment line is parsed as ` [] []`. - A bare `` preserves the previous behavior: harness only, with no model or effort launch flag. When the harness token is absent or `default`, secondmate launch falls back through `config/crew-harness` and then the primary's own harness, and no model or effort is read from that file. - `fm-harness.sh secondmate-model` and `fm-harness.sh secondmate-effort` expose only the optional tokens from `config/secondmate-harness`; `config/crew-harness` remains a bare adapter-name file. Changing this pin affects the next secondmate spawn or control-plane relaunch; the relaunch profile rules are owned by [`docs/agent-control.md`](agent-control.md#transactional-relaunch). - -### Per-launch overrides and inherited defaults - An explicit harness argument to `fm-spawn.sh` still overrides either config file for that spawn only. An explicit `--model` or `--effort` overrides the matching token from `config/secondmate-harness`; for a local route, an explicit harness or raw launch command starts with clean model and effort defaults unless those flags are also passed. - Remote secondmate routes accept verified harness adapters only and reject raw launch commands. When `config/crew-dispatch.json` exists, crewmate and scout spawns require an explicit resolved harness instead of automatically falling back to `config/crew-harness`. - The inherited-local-material contract is owned by [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md); its harness-relevant consequence is that a secondmate's own crewmates use the primary's dispatch profiles and static harness value. Those inherited values are defaults and rules only; `fm-spawn` still permits a consciously chosen explicit runtime outside the config. - `config/secondmate-harness` is not inherited because secondmates do not launch secondmates. - -### Installed hooks and launch details - For grok, `fm-spawn.sh` installs one firstmate-owned global turn-end hook under `$GROK_HOME/hooks/`, or `~/.grok/hooks/` when `GROK_HOME` is unset, and drops a per-task `.fm-grok-turnend` pointer in the worktree, with teardown removing the task token and pointer. For Kimi crews, `fm-spawn.sh` runs `fm-kimi-turnend-hook.sh install`, drops a per-task `.fm-kimi-turnend` pointer in the worktree, and records the matching private registry token for teardown. - Kimi continues to use the captain's normal Kimi home, including the existing config, skills, and memory; Firstmate does not create an isolated Kimi home. The Kimi installer requires an existing regular non-symlink `~/.kimi-code/config.toml`, `python3` with `tomllib`, and `jq`; it validates but never serializes the captain's TOML and refuses before writing when the config is missing, malformed, or surprising or when either tool requirement is unavailable. - Its `remove` action excises only the marker-delimited Firstmate region and removes Firstmate's hook files. For Pi and pi-signed secondmate launches, `fm-spawn.sh` starts the selected executable with `-e` pointed at the secondmate home's own tracked `.pi/extensions/fm-primary-pi-watch.ts` and `.pi/extensions/fm-primary-turnend-guard.ts`, both already present from the secondmate home's git worktree. - For omp secondmate launches, `fm-spawn.sh` passes no `-e` at all: omp auto-discovers the home's tracked `.omp/extensions/` with no trust gate, and naming a discovered file with `-e` as well loads it twice; every omp launch instead carries the tracked `.omp/fm-worker-overlay.yml` posture overlay through `--config`, which [`fm-spawn.sh --help`](../bin/fm-spawn.sh) owns. ## Claude permission mode (config/claude-permission-mode) -The optional local, gitignored `config/claude-permission-mode` selects the permission flag for every Claude worker launch: crewmates, scouts, Claude secondmates, and control-plane relaunches. - -### Accepted values and refusals - +The optional local, gitignored `config/claude-permission-mode` holds one token selecting the permission flag every Claude worker launch carries: crewmates, scouts, Claude secondmates, and control-plane relaunches alike. The token is the file's whitespace-trimmed content. - -| Token | Launch permission flag | -| --- | --- | -| `bypass` | `claude --dangerously-skip-permissions` | -| `auto` | `--permission-mode auto` | - -An absent file defaults to bypass, so an unconfigured home launches byte-for-byte as before. -Auto is Claude Code's classifier-reviewed permission mode, for a captain who refuses to run workers in bypass mode. -Only the permission flag changes. -The environment prefix, inline settings, model, effort flags, and every other part of the Claude launch stay unchanged. - -Any other value or an unreadable file refuses every spawn from that home, whichever harness it would launch. -This happens before any endpoint, worktree, or task record exists. -The diagnostic names the accepted values; Firstmate never falls back to a permission posture the captain did not choose. - -### When changes apply and inheritance - +`bypass` keeps today's launch, `claude --dangerously-skip-permissions`, and is also the default when the file is absent, so an unconfigured home launches byte-for-byte as before. +`auto` replaces that flag with `--permission-mode auto`, Claude Code's classifier-reviewed permission mode, for a captain who refuses to run workers in bypass mode; every other part of the Claude launch, including its environment prefix, inline settings, model, and effort flags, is unchanged. +Any other value, or an unreadable file, refuses every spawn from that home, whichever harness it would launch, before any endpoint, worktree, or task record exists, and names the accepted values; Firstmate never falls back to a permission posture the captain did not choose. `bin/fm-spawn.sh` reads the file on every spawn and relaunch, so a change takes effect at the next launch without a restart. The file is a captain-wide safety preference, so it is inherited into secondmate homes under the [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md) inherited-local-material contract; a secondmate's own Claude crewmates then launch on the same posture. - The [Claude adapter reference](../.agents/skills/harness-adapters/references/harness/claude.md) records the verified shape of both launches and which once-per-machine dialog each one can meet. -## Worker account pin (config/claude-account, config/pi-account) - -A home that mixes accounts for one runner, such as a work login and a personal one, can pin the account its own Claude and Pi workers launch on. -The pin is opt-in: with neither file, every launch is unchanged, and Claude workers keep receiving firstmate's own `CLAUDE_CONFIG_DIR` when it is set. - -Both files are local and gitignored. - -| Runner | File | Variable the launch receives | `ordinary` means | -| --- | --- | --- | --- | -| `claude` | `config/claude-account` | `CLAUDE_CONFIG_DIR` | the variable unset, so Claude uses its default login | -| `pi`, `pi-signed` | `config/pi-account` | `PI_CODING_AGENT_DIR` | `~/.pi/agent` | - -### File format and provider selection - -`config/claude-account` holds one line: `ordinary`, or the absolute path of an existing Claude config directory. -`config/pi-account` holds that same root on line 1 and, on line 2, the providers this home may spend, separated by spaces, for example `openai-codex anthropic`. - -A final newline is optional; any other line, a relative path, or a control character such as a CR refuses. -For Claude, `ordinary` unsets `CLAUDE_CONFIG_DIR` rather than pointing it at `~/.claude`, because Claude reads `$CLAUDE_CONFIG_DIR/.claude.json` and keys its macOS Keychain entry to any directory that is set ([authentication, "Credential management"](https://code.claude.com/docs/en/authentication#credential-management)). - -A Pi root can hold several provider logins at once, so the root alone does not say which account a launch spends. -A pinned Pi launch therefore needs `--model /` naming a declared provider, and Firstmate also passes `--provider ` so Pi cannot resolve the model under another signed-in provider. - -An unqualified model, an undeclared provider, or a raw Pi launch command, which cannot receive that flag, refuses; Firstmate never guesses a provider. - -### Launch scope and sign-in checks - -When a file is present, every launch of that runner from this home uses it: ships, scouts, local secondmate agents, raw Claude launch commands, and relaunches. -A raw Claude launch command refuses if its leading assignments set `CLAUDE_CONFIG_DIR` or a credential that a pinned launch unsets, such as `ANTHROPIC_API_KEY`. -The assignment would override the pin. -The refusal names the variable; remove that assignment from the raw command, or change or remove `config/claude-account`. - -Before any worker endpoint, local copy, or task record exists, and before a relaunch stops the running worker, Firstmate asks the runner itself whether the pinned account is signed in: `claude auth status` for Claude, and `pi auth check --provider --json --no-refresh` for Pi, falling back to `pi --list-models ` for a provider an extension registers. -The check runs with only `HOME`, `PATH`, `TMPDIR`, `USER`, `LOGNAME`, and the pinned root in its environment, so a credential variable in firstmate's own environment cannot answer for an empty root. - -A pinned Claude launch also unsets the environment credentials Claude ranks above a stored login, such as `ANTHROPIC_API_KEY`, `CLAUDE_CODE_OAUTH_TOKEN`, and the Bedrock and Vertex switches ([authentication precedence](https://code.claude.com/docs/en/authentication#authentication-precedence)). -Pi ranks a root's stored logins above environment variables, so a pinned Pi launch unsets nothing. - -A home that authenticates Claude through environment credentials on purpose should leave the pin absent. - -### Failures, reporting, and inheritance - -A malformed file, a root that is not a readable directory, or a signed-out account refuses the launch and names the file to fix; Firstmate never falls back to the ambient account and never changes a global login or copies a credential. -The spawn prints the pin as `account=` (plus `account_provider=` for Pi) and records the same fields in the task record, so the session-start digest shows which account each worker launched on. - -Pins are not inherited into secondmate homes: a local secondmate agent launches on the launching home's pin, while the secondmate's own workers read the secondmate home's files. -A remote secondmate is launched on its host from its own home's configuration, so create the file in that remote home. - -[`bin/fm-worker-account-lib.sh`](../bin/fm-worker-account-lib.sh) owns parsing, the sign-in check, and the full list of credentials a Claude launch unsets; [runtime backend verification](verification/runtime-backends.md#worker-account-pin-sign-in-check) records the check against the real runners. - ## Lavish server address (config/lavish-axi-host) The optional local, gitignored `config/lavish-axi-host` contains one non-empty address without whitespace for the per-machine Lavish server. `fm-spawn.sh` exports that address into every new worker and relaunch for opening boards, and the file is inherited into secondmate homes through the primary-authoritative configuration contract. - Once a board exists, the process-event adapter derives the polling address from that board's own saved Lavish session instead; its header owns the lookup contract. When the file is absent, worker launches do not add a board address and retain the existing ambient-environment behavior. - Malformed or unreadable values refuse the launch before the worker starts. The address selects the existing shared server; it does not authorize starting or stopping the server, and the Lavish startup crash remains a vendor-tool concern. ## Home brief include (config/brief-include.md) -The optional local, gitignored `config/brief-include.md` adds standing worker instructions to every ship and scout brief. -This keeps private brief content out of tracked files. +The optional local, gitignored `config/brief-include.md` carries standing worker instructions that one captain wants on every ship and scout brief, so private brief content needs no edit to a tracked file. When the file exists, `bin/fm-brief.sh` appends its text verbatim as the scaffold's last section, `# Home brief additions`, which defers to every other section of the brief, including the ship contract a later scout promotion appends below it. - An absent or blank file changes nothing, while a present path that is not a readable regular file, or text carrying its own `Delivery contract: mode=` line, stops the scaffold before anything is written. The text is static and never executed or expanded; secondmate charters never take it, and the file is local to each home rather than part of secondmate inherited configuration. - `bin/fm-brief.sh`'s header owns the placement rule and its safety argument. ## Worker launch environment (config/launch-env-allowlist) The optional local, gitignored `config/launch-env-allowlist` limits the ambient environment passed to newly launched workers, scouts, and secondmates, including relaunches. With no file, ambient inheritance remains unfiltered: selected harness markers are cleared, while the provider, long-lived terminal daemon, and shell initialization determine which other variables reach the worker. - Do not assume every worker inherits the invoking Firstmate process's current environment. The file is inherited into secondmate homes through the [primary-authoritative configuration contract](../.agents/skills/secondmate-provisioning/SKILL.md). - Changes apply to subsequent launches; existing processes keep their environment. -### Allowlist format - Create the file with one environment variable **name** per line, never credential values, assignments, wildcards, or shell commands. Blank lines and lines beginning with `#` are allowed. - Invalid names, an unreadable or nonregular file, or a path inspection error (including an inaccessible configuration directory) stop the launch. An empty file enables filtering with only Firstmate's operational floor. - For example, a provider using `OPENAI_API_KEY` and Git using an SSH agent could use: ```text @@ -918,19 +422,13 @@ OPENAI_API_KEY SSH_AUTH_SOCK ``` -### Variables retained and where values come from - Firstmate retains basic home, executable search, terminal, locale, temporary-directory, and backend routing variables, plus its explicit launch assignments, its ship and scout task marker, the compact-adviser kill switch described below, and enabled task trace. [`fm-spawn.sh --help`](../bin/fm-spawn.sh) owns the exact retained names and parsing mechanics. - Other ambient names must be listed explicitly, including custom credential-store locations, proxy settings, and certificate overrides when required by the selected tools. The command shell and worker may still create their own variables. - Allowed values come from the destination pane at execution time; they are neither copied from the invoking Firstmate process nor written into the launch command. Listing a name does not provision it in a daemon's environment or transfer credentials to another machine. -### Authentication requirements - Choose the minimum additions for the authentication method actually in use: | Provider or Git transport | Additional names needed | @@ -943,46 +441,27 @@ Choose the minimum additions for the authentication method actually in use: | Git over SSH with a key file | No credential variable when normal SSH configuration selects the key; file permissions and any passphrase handling still apply. | | Git over HTTPS with a credential helper | Whatever the configured helper requires; a GitHub CLI helper using an environment token needs its selected `GH_TOKEN` or `GITHUB_TOKEN`. | -### Validation and security limits - Verify the selected provider login and Git transport after opting in; Firstmate does not infer credentials from model names or install a secret manager. Raw launch commands run under noninteractive POSIX `sh` with this option and must use compatible syntax. - The filter runs at the worker command boundary, after the terminal daemon and pane shell have started; it does not scrub either of those processes. This is not a sandbox: it cannot revoke same-user access to credential files, prevent tools or later shells from loading credentials again, or isolate processes from the same user's other processes. - Regression coverage executes emitted launch commands with synthetic nonsecret values in [`tests/fm-spawn-dispatch-profile.test.sh`](../tests/fm-spawn-dispatch-profile.test.sh). -### Compact adviser setting - Every crewmate, scout, and secondmate Firstmate launches starts with `COMPACT_ADVISER_DISABLE=1` in its environment, on a fresh spawn and on a relaunch alike, so an unattended session never activates the compact adviser. This guarantee also covers raw launch commands, remote secondmates, and launches filtered by `config/launch-env-allowlist`; it does not depend on the destination environment already containing the variable. - Firstmate provides no configuration or flag to change this value. This applies only to agents Firstmate launches; the captain's own primary Firstmate session is never given the variable. - [`fm-spawn.sh --help`](../bin/fm-spawn.sh) owns the delivery mechanics, with focused regression coverage in [`tests/fm-spawn-compact-adviser-disable.test.sh`](../tests/fm-spawn-compact-adviser-disable.test.sh) and [`tests/fm-spawn-compact-adviser-disable-remote.test.sh`](../tests/fm-spawn-compact-adviser-disable-remote.test.sh). Every claude launch's inline `--settings` JSON also carries `"attribution":{"commit":"","pr":"","sessionUrl":false}`, so a spawned worker never writes a Co-Authored-By trailer, Claude-Session link, or generated-with line into a commit or PR body regardless of which settings scopes end up loaded. -Every fleet launch, Claude included, also receives a pane-scoped `GIT_CONFIG` `core.hooksPath` pointing at `state/.git-hooks`, so git's `commit-msg` hook strips known AI trailers at the commit object even when a runtime injects them after the typed message. -`bin/fm-git-strip-ai-trailers.sh` owns the identities, the install, and chaining the hooks of whichever repository git is running in, so a project hook such as husky still runs. -That directory is read-only, so a hook manager run inside a fleet pane (lefthook's npm postinstall, `pre-commit install`) fails instead of displacing the strip; install a project's hooks from outside the pane, where the wrappers chain them. -Per-machine Cursor `cli-config.json` attribution-off is not this contract: it does not travel with Firstmate, defaults back to on when unset, and only feeds the CLI's request to the server, so it suppresses the trailer rather than preventing it. ## Crew dispatch profiles (config/crew-dispatch.json) `config/crew-dispatch.json` is an optional local, gitignored file containing natural-language rules that firstmate reads before dispatching a crewmate or scout. -Firstmate chooses the best matching rule with judgment; shell scripts do not match the natural-language rules. -Firstmate resolves the rule's profile object or array under `AGENTS.md` section 4 and `quota-array-dispatch`, then passes only concrete `--harness`, `--model`, and `--effort` flags to `fm-spawn.sh`. - -**Spawn requirements** - -- When the file exists, `fm-spawn.sh` enforces that contract by refusing crewmate and scout spawns that lack an explicit harness (`--harness`, a positional adapter, or a raw launch command). -- Batch spawns satisfy the same requirement with a shared `--harness`. -- Secondmate spawns are exempt and still resolve through `config/secondmate-harness` and its optional model and effort tokens. - -**Contract owners** - +The shell scripts do not match those rules; firstmate chooses the best matching rule with judgment, resolves its profile object or array under the operating contract in `AGENTS.md` section 4 and `quota-array-dispatch`, and passes only concrete `--harness`, `--model`, and `--effort` flags to `fm-spawn.sh`. +When the file exists, `fm-spawn.sh` enforces that contract by refusing crewmate and scout spawns that lack an explicit harness (`--harness`, a positional adapter, or a raw launch command). +Batch spawns satisfy the same requirement with a shared `--harness`. +Secondmate spawns are exempt and still resolve through `config/secondmate-harness` and its optional model and effort tokens. This section is the single owner of the canonical schema and its per-field semantics. `AGENTS.md` section 4 owns the always-loaded dispatch intake boundary, and `quota-array-dispatch` owns the completion-aware profile-array selection procedure. @@ -992,7 +471,6 @@ This section is the single owner of the canonical schema and its per-field seman { "when": "", "approval": "captain", - "min_confidence": 0.85, "floor": { "scope": "", "min_percent": 20, "provider": "" }, "use": [ { "harness": "", "model": "", "effort": "", "provider": "", "floor": { "scope": "", "min_percent": 50 } } @@ -1006,273 +484,121 @@ This section is the single owner of the canonical schema and its per-field seman } ``` -**Required and optional fields** - -| Field | Requirement | -| --- | --- | -| `rules` | May be absent or empty for a default-only configuration. | -| Rule `when` and `use` | Required for each rule. | -| `use` and optional top-level `default` | Accept one profile object or a non-empty array of profile objects; the single-object form remains fully backward-compatible. | -| Profile `harness` | Required in every profile. | -| Profile `model` and `effort`; rule `why` | Optional. | - -**Fields applied only by typed resolution** - -Rule `approval`, `min_confidence`, and `floor`, and profile `provider` and `floor` are optional declarations that only [typed dispatch resolution](#typed-dispatch-resolution-env-typesafe_api_key) applies in code; without that opt-in they are inert, and firstmate's own intake reads them as ordinary hints. +Per rule, `when` and `use` are required; the top-level `rules` array itself may be absent or empty for a default-only configuration. +Both `use` and the optional top-level `default` accept either one profile object or a non-empty array of profile objects. +The single-object form stays fully backward-compatible, and every profile needs `harness`. +Profile `model` and `effort` fields and rule `why` are optional. +Rule `approval` and `floor`, and profile `provider` and `floor` are optional declarations that only [typed dispatch resolution](#typed-dispatch-resolution-env-typesafe_api_key) applies in code; without that opt-in they are inert, and firstmate's own intake reads them as ordinary hints. The resolver supplies the fixed neutral Choice option `No listed rule applies to this task.` for work that matches no listed rule. - -- `approval` accepts only `"captain"` and means a task the rule matches is never dispatched from the tool's answer alone. - -`min_confidence` is a number from 0 through 1. -The rule's own probability in the answer must reach it, replacing the resolver's global 0.6 floor on the answer's confidence. -Set it high when a wrong pick is costly and low when the rule is a safe runner-up. - -**Rule quota floors** - -- A rule `floor` names the quota-axi `provider` and `scope` whose `effectivePercentRemaining` must be at least `min_percent` for the rule's profiles to apply. -- A provider-only rule floor on an expanded provider binds to its `default` account row. -- An absent or unknown row or unmeasured provider makes the floor unverifiable and escalates without authorizing default routing. -- A known percentage below the floor makes the tool resolve among `default` profiles instead. - -**Provider identifiers and mappings** - +`approval` accepts only `"captain"` and means a task the rule matches is never dispatched from the tool's answer alone. +A rule `floor` names the quota-axi `provider` and `scope` whose `effectivePercentRemaining` must be at least `min_percent` for the rule's profiles to apply. +A provider-only rule floor on an expanded provider binds to its `default` account row. +An absent or unknown row or unmeasured provider makes the floor unverifiable and escalates without authorizing default routing. +A known percentage below the floor makes the tool resolve among `default` profiles instead. A profile `provider` optionally names the quota-axi provider family whose rows apply to that profile; when present, profile and rule-floor provider IDs must match the strict whole-string pattern `^[a-z0-9]+(-[a-z0-9]+)*\z`. -Bootstrap validates resolver-only `approval`, `min_confidence`, `floor`, and present `provider` values only while typed resolution is active; without the key those inert fields and the pre-existing verified-harness baseline preserve bootstrap behavior. - +Bootstrap validates resolver-only `approval`, `floor`, and present `provider` values only while typed resolution is active; without the key those inert fields and the pre-existing verified-harness baseline preserve bootstrap behavior. Typed resolution additively recognizes `gemini` because AGENTS.md section 4 verifies it for crewmate and scout dispatch. - -| Harness | Provider declaration on the opted-in resolver path | -| --- | --- | -| `claude`, `codex`, `grok`, `kimi`, `cursor`, `agy`, `muse` | The resolver has an authoritative single-provider mapping. | -| Every other verified harness | Must declare `provider` explicitly; this includes multi-provider `pi`, `pi-signed`, `omp`, and `opencode`, and unmapped `gemini`, `rovo`, and `devin`; omission is an actionable configuration error before any request. | - -This single-provider table is separate from the frozen legacy mapping used by `fm-quota-choose.sh`, so additions cannot alter no-key routing. - -**Profile quota floors** - -- A profile `floor` contains only `scope` and `min_percent`, always uses that profile's provider and matched account, and makes that one candidate ineligible below `min_percent` on the named scope. -- An absent or unknown named row also makes the candidate unrankable and is reported as an unverifiable floor, not as a known shortfall. - -**Model, effort, and fallback behavior** - -- `ultra` is native-only: the model-aware validation contract and launch mapping are owned by `bin/fm-harness.sh validate-native-effort` and `bin/fm-spawn.sh` respectively. -- Codex `max` is valid when the profile selects `gpt-5.6-luna`, whose installed catalog entry supports that reasoning level. -- An omitted model or effort means the selected harness uses its own default for that axis. -- Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. -- If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. -- Except for `ultra`, which refuses unsupported profiles under the native-effort contract above, an effort value the chosen harness does not accept is recorded as `effort=` in task meta for traceability but omitted from the launch flags. -- Bootstrap reports unsupported harness/model/effort combinations as a `CREW_DISPATCH` diagnostic when they are visible in the file. - +The opted-in resolver has authoritative single-provider mappings for `claude`, `codex`, `grok`, `kimi`, `cursor`, `agy`, and `muse`; every other verified harness must declare `provider` explicitly, including multi-provider `pi`, `pi-signed`, `omp`, and `opencode` and unmapped `gemini`, `rovo`, and `devin`. +Its single-provider table is separate from the frozen legacy mapping used by `fm-quota-choose.sh`, so additions cannot alter no-key routing. +The resolver returns an actionable configuration error before any request when such a profile omits it. +A profile `floor` contains only `scope` and `min_percent`, always uses that profile's provider and matched account, and makes that one candidate ineligible below `min_percent` on the named scope. +An absent or unknown named row also makes the candidate unrankable and is reported as an unverifiable floor, not as a known shortfall. +`ultra` is native-only: the model-aware validation contract and launch mapping are owned by `bin/fm-harness.sh validate-native-effort` and `bin/fm-spawn.sh` respectively. +Codex `max` is valid when the profile selects `gpt-5.6-luna`, whose installed catalog entry supports that reasoning level. +An omitted model or effort means the selected harness uses its own default for that axis. +Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. +If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. +Except for `ultra`, which refuses unsupported profiles under the native-effort contract above, an effort value the chosen harness does not accept is recorded as `effort=` in task meta for traceability but omitted from the launch flags. +Bootstrap reports unsupported harness/model/effort combinations as a `CREW_DISPATCH` diagnostic when they are visible in the file. See [`docs/examples/crew-dispatch.json`](examples/crew-dispatch.json) for a starting point to copy into local `config/crew-dispatch.json`; its Pi default declares the `claude` provider required for typed resolution of that Anthropic model. - -**Validation and diagnostics** - -- When the file exists, bootstrap validates it with `jq`. -- Valid files stay silent by default; with `FM_BOOTSTRAP_VERBOSE_FACTS=1`, bootstrap emits `BOOTSTRAP_INFO: crew dispatch active config/crew-dispatch.json`, one `BOOTSTRAP_INFO:` fact per rule, and one fact for the optional default profile set. -- Malformed JSON, malformed rules, an empty or malformed profile array, an unverified harness, or an effort value unsupported by that harness is reported as `CREW_DISPATCH: invalid config/crew-dispatch.json - ...`. -- While typed resolution is active, malformed `approval`, `min_confidence`, `floor`, and present `provider` declarations receive the same diagnostic; without the key those inert declarations preserve the pre-existing bootstrap behavior. -- Missing `jq` is reported through the normal `MISSING: jq` install-consent flow. -- While the file remains present, no crewmate or scout spawn may proceed without an explicit resolved harness; malformed configuration must be reported and corrected rather than selected around. - -**Inheritance** - +When the file exists, bootstrap validates it with `jq`. +Valid files stay silent by default; with `FM_BOOTSTRAP_VERBOSE_FACTS=1`, bootstrap emits `BOOTSTRAP_INFO: crew dispatch active config/crew-dispatch.json`, one `BOOTSTRAP_INFO:` fact per rule, and one fact for the optional default profile set. +Malformed JSON, malformed rules, an empty or malformed profile array, an unverified harness, or an effort value unsupported by that harness is reported as `CREW_DISPATCH: invalid config/crew-dispatch.json - ...`. +While typed resolution is active, malformed `approval`, `floor`, and present `provider` declarations receive the same diagnostic; without the key those inert declarations preserve the pre-existing bootstrap behavior. +Missing `jq` is reported through the normal `MISSING: jq` install-consent flow. +While the file remains present, no crewmate or scout spawn may proceed without an explicit resolved harness; malformed configuration must be reported and corrected rather than selected around. Secondmate homes inherit this file from the primary, so a secondmate's own crewmates apply the same dispatch profile behavior. ## Typed dispatch resolution (.env TYPESAFE_API_KEY) `bin/fm-dispatch-resolve.sh` resolves one concrete crewmate or scout profile from a written brief with typesafe.ai's System One model (Jev), so the rule match that firstmate otherwise reasons out in its own context becomes one short tool turn. It is off unless `TYPESAFE_API_KEY` is non-empty in the calling environment or the home's gitignored `.env` holds a `TYPESAFE_API_KEY=` line; the environment wins, matching the Relay and mail-plane contracts, and the Relay accessor in `bin/fm-env-lib.sh` reads the line. - Off means one `dispatch-resolve: off` line on stderr, nothing on stdout, exit 0, and no network call, so firstmate dispatches exactly as it does without the tool. This section is the single owner of the tool's operator contract; the script header owns its exact flags and output lines, and "Crew dispatch profiles" above owns the declared rule and profile fields it applies. - Rules come only from the effective home's `config/crew-dispatch.json`; `FM_CONFIG_OVERRIDE` selects the config directory for tests and specialized setup like the other scripts. ```sh bin/fm-dispatch-resolve.sh data//brief.md --project # TOON block on stdout ``` -**When firstmate invokes the resolver** - Firstmate invokes the resolve path directly after writing the brief, without a preflight; the absent-key off line is handled exactly like every other non-clear outcome. - -**What the model receives** - -When on and at least one rule exists, the tool sends the project name and the brief's task-specific text as state and asks one Choice question whose options are every rule's `when` plus the fixed neutral option for no matching rule; the model never sees quota, catalogs, `why`, `use`, approvals, or confidence floors. -The task-specific text is the brief's `## Captain's intent` and `## Firstmate spec` sections under `# Task` that `bin/fm-brief.sh` scaffolds, read by the same parser that feeds `fm-spawn.sh` validation and the no-mistakes `--intent` contract; a brief with neither section is sent whole. - -When the sections are sent from a scout brief, the line `Brief kind: scout (report only)` comes first, taken from the scaffold's scout contract line; ship briefs and briefs sent whole get no kind line. -A ship brief's delivery mode is deliberately not sent, because in live runs naming it pushed a routine ship brief toward the hardest tier (see [the verification record](verification/dispatch-resolve.md)). - -The scaffold's standard setup, rules, and definition-of-done text is the same in every brief, so leaving it out keeps its safety language from reading as a signal about the task. - -**Missing or invalid rules** - +When on and at least one rule exists, the tool sends the project name and the whole brief as state and asks one Choice question whose options are every rule's `when` plus the fixed neutral option for no matching rule; the model never sees quota, catalogs, `why`, `use`, or approvals. An absent rules file, a default-only file, or `rules: []` returns the non-clear reason `no rules to match` without a model or quota request, leaving firstmate's existing routing in control; an existing but unreadable or malformed rules file, including a broken symlink, remains an actionable exit 2 configuration error. - -**Checks performed after the answer** - -After the answer, code applies all remaining checks and ranking: - -- The confidence floor and the matched rule's `approval` and `floor`. -- Each candidate's `provider` and `floor`. -- Every applicable account-wide and model/product row from one `quota-axi --json` snapshot. -- The numeric `spendPriority` argmax over candidates, using each candidate's limiting row. - +Everything after the answer runs in code: the confidence floor, the matched rule's `approval` and `floor`, each candidate's `provider` and `floor`, every applicable account-wide and model/product row from one `quota-axi --json` snapshot, and the numeric `spendPriority` argmax over candidates using each candidate's limiting row. The [shared quota library](../bin/fm-quota-axi-lib.sh) accepts schema 5 and schema 6 and implements the [account-matching contract](../.agents/skills/quota-array-dispatch/SKILL.md#1-eligibility). - -- An expanded provider with no matching account row leaves the candidate eligible but unranked. -- Known applicable rows from a provider with partial quota semantics remain rankable; rows whose own status is not known remain unrankable. - -**Confidence and fallback rules** - -- A rule that declares `min_confidence` is checked against that rule's own probability, whether it is the picked option or a runner-up, so a runner-up never needs weaker support than it would as the pick. -- A picked rule without `min_confidence`, and the neutral option, keep the global 0.6 floor on the answer's confidence exactly as before, so a file with no declared floors behaves as it did. - -When the picked rule declares its own floor but its probability falls below it, the tool checks the other options: - -1. Find the most probable other option that clears its own floor: the rule's `min_confidence`, or 0.6 otherwise. -2. Print a `fallback:` line naming both floors and resolve that rule as though it had been picked. - -No qualifying option, or two equally probable qualifying options, produces `ambiguous`. - -**Candidate eligibility and evidence** - -- Any applicable `exhausted_now` row or known zero bound makes that candidate ineligible, and a known profile-floor shortfall does the same before unrelated quota uncertainty is considered. -- Missing or nonnumeric `spendPriority` evidence is never ranked, and every candidate is printed beside its evidence or the reason it was not rankable, including on ambiguous and approval-gated outcomes that emit no profile. -- On the opted-in path, duplicate concrete profiles with the same harness, model, and effort inside one rule or the default array are configuration errors rather than ties. - -**Outcomes and exit status** - -| Result | Meaning | -| --- | --- | -| `clear` | A `profile:` line ready for `fm-spawn.sh`. | -| `ambiguous` | Confidence below the floor with no runner-up taken. | -| `escalate` | An approval-gated rule, unverifiable rule floor, nothing rankable, or a genuine tie. | -| `error` | API, network, malformed response metadata, rendering, or quota-axi failure. | - -Every result above exits 0. - -- Response probabilities must contain exactly every offered choice, use numeric values from 0 through 1, and sum to approximately 1 within 0.01. -- Only a usage or configuration error exits 2: an unreadable brief, an existing but unreadable or malformed canonical rules file, or missing `jq`, each reported and never selected around. -- Missing `curl` is a normal structured `error` outcome with exit 0 so firstmate uses today's routing. - -**Firstmate retains the dispatch decision** - +An expanded provider with no matching account row leaves the candidate eligible but unranked. +Known applicable rows from a provider with partial quota semantics remain rankable; rows whose own status is not known remain unrankable. +Any applicable `exhausted_now` row or known zero bound makes that candidate ineligible, and a known profile-floor shortfall does the same before unrelated quota uncertainty is considered. +Missing or nonnumeric `spendPriority` evidence is never ranked, and every candidate is printed beside its evidence or the reason it was not rankable, including on ambiguous and approval-gated outcomes that emit no profile. +On the opted-in path, duplicate concrete profiles with the same harness, model, and effort inside one rule or the default array are configuration errors rather than ties. +The result is one of `clear` (a `profile:` line ready for `fm-spawn.sh`), `ambiguous` (confidence below the floor), `escalate` (an approval-gated rule, unverifiable rule floor, nothing rankable, or a genuine tie), or `error` (API, network, malformed response metadata, rendering, or quota-axi failure), and every one of them exits 0. +Response probabilities must contain exactly every offered choice, use numeric values from 0 through 1, and sum to approximately 1 within 0.01. +Only a usage or configuration error exits 2: an unreadable brief, an existing but unreadable or malformed canonical rules file, or missing `jq`, each reported and never selected around. +Missing `curl` is a normal structured `error` outcome with exit 0 so firstmate uses today's routing. The tool never replaces firstmate's judgment, `quota-array-dispatch`, the captain-approval gate, or `fm-spawn.sh` validation; `AGENTS.md` section 4 owns what firstmate does with each outcome. By accepted design, a `clear` result does not enforce catalog/authentication, reasoning-class, or completion-runway gates. - Firstmate passes its profile line unless it states a reason to override, such as the brief's reasoning class or an eligible-unranked-candidate note; every non-clear result returns to the full existing intake. -**Key handling and fixed settings** - -- The resolver and bootstrap copy an environment-provided key into a non-exported private variable and unset `TYPESAFE_API_KEY` before launching child processes, so the secret is absent from child environments. -- The resolver sends the key to `curl` only as a header read from a file descriptor, never on argv, and nothing prints, logs, or writes it. -- The resolver fixes the endpoint at `https://api.typesafe.ai`, model at `jev-latest`, default confidence floor at 0.6, and request timeout at 5 seconds; `TYPESAFE_API_KEY` is its only resolver-specific environment setting. - +The resolver and bootstrap copy an environment-provided key into a non-exported private variable and unset `TYPESAFE_API_KEY` before launching child processes, so the secret is absent from child environments. +The resolver sends the key to `curl` only as a header read from a file descriptor, never on argv, and nothing prints, logs, or writes it. +The resolver fixes the endpoint at `https://api.typesafe.ai`, model at `jev-latest`, confidence floor at 0.6, and request timeout at 5 seconds; `TYPESAFE_API_KEY` is its only resolver-specific environment setting. The live rule-match evidence is recorded in [`verification/dispatch-resolve.md`](verification/dispatch-resolve.md). ## Toolchain On session start the first mate detects what its required toolchain is missing or too old and lists each problem with either an exact install command or manual instructions. It installs automatically supported tools only after you say go; manual-only tools remain for you to install from the printed instructions. - Required tools come in two parts: a universal toolchain every home needs regardless of backend, and a per-backend delta that follows the runtime backend actually resolved for this home. - -**Universal requirements** - -Every home requires: - -- node and git. -- gh, with GitHub authentication through `gh auth login`. -- no-mistakes v1.46.0 or newer. -- Compatible gh-axi. -- chrome-devtools-axi. -- Compatible tasks-axi, as specified in "Backlog backend" above. -- Compatible quota-axi. - +The essential universal toolchain is node, git, gh with GitHub auth via `gh auth login`, no-mistakes v1.46.0 or newer, compatible gh-axi, chrome-devtools-axi, compatible tasks-axi per "Backlog backend" above, and compatible quota-axi. [`bin/fm-bootstrap.sh`](../bin/fm-bootstrap.sh) owns the axi-family floor policy and the gh-axi and lavish-axi floors, while [`bin/fm-tasks-axi-lib.sh`](../bin/fm-tasks-axi-lib.sh) and [`bin/fm-quota-axi-lib.sh`](../bin/fm-quota-axi-lib.sh) hold their own tools' floor constants. This section is the single owner of that universal toolchain list; backend guides' prerequisites point here and add only their backend-specific tools. - In that list, no-mistakes runs the validation pipeline, gh-axi and chrome-devtools-axi cover GitHub and browser operations, and tasks-axi plus quota-axi back backlog mutations and quota-aware array dispatch. Lavish is a presentation-only dependency for visual decisions and reports; nonvisual work can proceed with plain text when it is unavailable. - -**Backend requirements** - The per-backend delta is required only for the backend resolved from `FM_BACKEND`, then `config/backend`, then runtime auto-detection, then default `tmux`, so a home is never told to install a tool an inactive backend or feature would need. -`fm_backend_required_tools` in `bin/fm-backend.sh` owns the backend additions: - -| Resolved backend | Additional tools | -| --- | --- | -| `tmux` | `tmux`, `treehouse` | -| `herdr` | `herdr`, `jq`, `treehouse` | -| `zellij` | `zellij`, `jq`, `treehouse` | -| `orca` | `orca` | -| `cmux` | `cmux`, `jq`, `treehouse` | - -The JSON-emitting adapters (`herdr`, `zellij`, `cmux`) need `jq` because their spawn and liveness paths parse backend JSON. -Every session-provider-only backend (`tmux`, `herdr`, `zellij`, `cmux`) uses `treehouse` for worktrees. - +That delta is owned in code by `fm_backend_required_tools` in `bin/fm-backend.sh`: the resolved backend's own session-provider CLI (`tmux`, `herdr`, `zellij`, `orca`, or `cmux`), `jq` for the JSON-emitting adapters (`herdr`, `zellij`, `cmux`) whose spawn and liveness paths parse the backend's JSON output, and the `treehouse` worktree provider for every session-provider-only backend (`tmux`, `herdr`, `zellij`, `cmux`). Backend tool availability uses the adapter's own executable resolver, so bootstrap and spawn agree on supported non-`PATH` locations such as cmux's bundled CLI. An unknown resolved backend emits `BACKEND_INVALID` and blocks dispatch instead of silently dropping its dependency delta or falling back to tmux. - Orca provides both the task worktree and terminal endpoint (see "Runtime backend" above), so `backend=orca` requires only `orca` on top of the universal toolchain and skips both `treehouse` and every other backend's session CLI. A herdr, zellij, or cmux home is therefore never told `tmux` is missing, and the `treehouse` durable-lease upgrade check runs only for the backends that actually use treehouse. - -**Feature-specific requirements** - -- When `config/crew-dispatch.json` exists, bootstrap also requires `jq` for dispatch profile validation. -- When Relay is opted in, bootstrap also requires `curl` and `jq` before arming the relay poll shim. - -**Missing-tool diagnostics** - +When `config/crew-dispatch.json` exists, bootstrap also requires `jq` for dispatch profile validation. +When Relay is opted in, bootstrap also requires `curl` and `jq` before arming the relay poll shim. `tasks-axi` and `quota-axi` are essential bootstrap tools in every profile. - -- An absent or incompatible `tasks-axi` reports `MISSING: tasks-axi (install: npm install -g tasks-axi)`; when `config/backlog-backend` is not `manual`, a home with a configured non-markdown adapter or a markdown backlog refuses lifecycle mutation until compatible `tasks-axi` is on `PATH`, while a manual-backend home keeps its backlog hand-edited. -- An absent or incompatible `gh-axi` reports `MISSING: gh-axi (install: npm install -g gh-axi && gh-axi setup hooks)`. -- An absent or incompatible `lavish-axi` reports `PRESENTATION_UNAVAILABLE` with its required floor, install command, and explicit text fallback; [`bootstrap-diagnostics`](../.agents/skills/bootstrap-diagnostics/SKILL.md) owns the response and compatibility check before visual use. -- An absent or too-old `quota-axi` reports `MISSING: quota-axi (install: npm install -g quota-axi)`; firstmate cannot resolve a profile array without a compatible binary. - -**Checkout diagnostics** - +An absent or incompatible `tasks-axi` reports `MISSING: tasks-axi (install: npm install -g tasks-axi)`; when `config/backlog-backend` is not `manual`, a home with a configured non-markdown adapter or a markdown backlog refuses lifecycle mutation until compatible `tasks-axi` is on `PATH`, while a manual-backend home keeps its backlog hand-edited. +An absent or incompatible `gh-axi` reports `MISSING: gh-axi (install: npm install -g gh-axi && gh-axi setup hooks)`. +An absent or incompatible `lavish-axi` reports `PRESENTATION_UNAVAILABLE` with its required floor, install command, and explicit text fallback; [`bootstrap-diagnostics`](../.agents/skills/bootstrap-diagnostics/SKILL.md) owns the response and compatibility check before visual use. +An absent or too-old `quota-axi` reports `MISSING: quota-axi (install: npm install -g quota-axi)`; firstmate cannot resolve a profile array without a compatible binary. Bootstrap also reports a `TANGLE:` line when `FM_ROOT` is on a named non-default branch; follow the printed checkout remediation rather than treating it as an installable tool problem. In a read-only session that did not get the fleet lock, the same line is advisory and omits the checkout command. - -**Project refresh at startup** - The locked session-start deferred network stage runs bootstrap's best-effort project clone refresh through `fm-fleet-sync.sh`; [`fm-bootstrap.sh`'s header](../bin/fm-bootstrap.sh) owns the exact clone-refresh overlap, liveness-before-convergence, per-mate concurrency, ordered diagnostic replay, and sequential-fallback contract. - -- It emits `FLEET_SYNC:` for skipped refreshes that may matter, recovered self-heals, and `STUCK:` alarms. -- Normal completed runs keep local-only and no-origin skips silent. -- If bootstrap kills a timed-out refresh, it replays any completed `fm-fleet-sync.sh` output before the aggregate timeout skip so no finished result is lost. - -**Stale Git lock recovery** - +It emits `FLEET_SYNC:` for skipped refreshes that may matter, recovered self-heals, and `STUCK:` alarms. +Normal completed runs keep local-only and no-origin skips silent. +If bootstrap kills a timed-out refresh, it replays any completed `fm-fleet-sync.sh` output before the aggregate timeout skip so no finished result is lost. A killed refresh (or a teardown process kill) can leave an orphaned `.git/packed-refs.lock` in a clone, which makes the next refresh's fetch fail with Git's `Unable to create '...packed-refs.lock': File exists`. On that signature only, `fm-fleet-sync.sh` retries the fetch with a bounded wait for the lock to self-clear, then removes the lock and retries once more only when it can prove the lock stale, exactly like the `fm-teardown.sh` `index.lock` recovery. - It never removes a live lock, leaves any other failure shape untouched, and prints every wait, retry, and removal to stderr plus a one-line `recovered:` summary to stdout on success so that this session-start relay still surfaces the recovery. - -**Secondmate sync at startup** - The same deferred network stage performs guarded tracked-file sync and propagates declared inherited local material into each validated live home under that sequencing contract. Local routes use direct guarded filesystem operations, while remote routes delegate sync and allowlisted transfer through their configured SSH host without probing any unconfigured fleet. - -- It emits `SECONDMATE_SYNC:` only when a home was skipped for an actionable sync reason, inheritance failed, or a divergent shared captain-preference copy was quarantined. -- When a running home advances and its loaded instruction surface (`AGENTS.md`, `bin/`, or `.agents/skills/`) changed, bootstrap sends the re-read nudge itself through the stable `fm-` selector and reports the exact completed send as `BOOTSTRAP_INFO:`. -- If that send fails, bootstrap keeps an idempotent retry marker and emits `NUDGE_SECONDMATES:` with the failure reason. -- The same bootstrap run emits `SECONDMATE_LIVENESS:` only when a registered secondmate is skipped or its relaunch fails; already-live and successfully relaunched secondmates are handled silently. - -**Push inherited configuration during a session** - +It emits `SECONDMATE_SYNC:` only when a home was skipped for an actionable sync reason, inheritance failed, or a divergent shared captain-preference copy was quarantined. +When a running home advances and its loaded instruction surface (`AGENTS.md`, `bin/`, or `.agents/skills/`) changed, bootstrap sends the re-read nudge itself through the stable `fm-` selector and reports the exact completed send as `BOOTSTRAP_INFO:`. +If that send fails, bootstrap keeps an idempotent retry marker and emits `NUDGE_SECONDMATES:` with the failure reason. +The same bootstrap run emits `SECONDMATE_LIVENESS:` only when a registered secondmate is skipped or its relaunch fails; already-live and successfully relaunched secondmates are handled silently. For a mid-session inherited local-material edit where tracked-file sync is not needed, run `bin/fm-config-push.sh`. It uses the same live secondmate discovery and propagation helper as bootstrap; its [help](../bin/fm-config-push.sh) owns reporting and exit semantics, and [`fm_config_inherit_items`](../bin/fm-config-inherit-lib.sh) declares the inherited items. - -- When an allowlisted config item changes for an already-running local home, it sends the literal-content reread pointer described in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md); unchanged allowlisted config sends no pointer unless a previous delivery is pending. -- A changed remote home instead receives one durably recorded marked re-read instruction after the allowlisted bytes have transferred because primary-local generation paths are not meaningful on another host. -- The locked bootstrap inheritance pass uses the same placement-specific behavior; see `secondmate-provisioning` for the single contract owner. -- That live discovery starts from `state/*.meta` records with `kind=secondmate`; `data/secondmates.md` only backfills `home=` for older or incomplete meta records. -- Skipped items, such as a destination checkout that does not yet gitignore the item, are visible warnings but not hard failures. +When an allowlisted config item changes for an already-running local home, it sends the literal-content reread pointer described in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md); unchanged allowlisted config sends no pointer unless a previous delivery is pending. +A changed remote home instead receives one durably recorded marked re-read instruction after the allowlisted bytes have transferred because primary-local generation paths are not meaningful on another host. +The locked bootstrap inheritance pass uses the same placement-specific behavior; see `secondmate-provisioning` for the single contract owner. +That live discovery starts from `state/*.meta` records with `kind=secondmate`; `data/secondmates.md` only backfills `home=` for older or incomplete meta records. +Skipped items, such as a destination checkout that does not yet gitignore the item, are visible warnings but not hard failures. ## Watched tool updates (config/watched-tools.json) @@ -1284,7 +610,6 @@ When it is present and the check is armed, [`bin/fm-tool-update-check.sh`](../bi The second condition is the reason the check exists. An update can install correctly and stay inert because an earlier `PATH` entry still holds an older copy, and a check that only asks whether a newer version is published reports that host as up to date. - The script therefore runs every copy of a watched command found on `PATH` and asks it for its own version, rather than trusting one lookup or reading a version out of a directory name. It only reports; it never installs, updates, fetches, or changes `PATH`, a version manager, or any installed tool. @@ -1310,63 +635,39 @@ This section is the single owner of the canonical schema. } ``` -**Entry fields and probe behavior** - -- Each entry needs a `name` and at least one of `command` or `git`; an entry may carry both. -- A `command` entry gives the `PATH` comparison above, and adding `announce_pattern` also reports the tool's own update announcement, which is how a tool that already reports its own updates is read rather than reimplemented. -- A tool does not always announce a new release on the command that prints its version: `no-mistakes --version` prints only the version, while its other commands carry the announcement. -- `announce_args` names the command to search for the announcement in that case, and it is asked only of the copy `PATH` resolves; without it the version probe's own output is searched. -- An `announce_pattern` that is not a usable extended regular expression stops `arm`, and during a sweep it is reported as that one tool's own check failure so one broken pattern never stops the other watched tools from being checked. -- A `git` entry reports how many commits the local clone is behind its remote branch, and stays silent when the clone is current or ahead. -- An omitted `branch` uses the remote's default branch, taken from the clone's own record of it and otherwise asked of the remote directly, so a `--single-branch` clone still resolves. - +Each entry needs a `name` and at least one of `command` or `git`; an entry may carry both. +A `command` entry gives the `PATH` comparison above, and adding `announce_pattern` also reports the tool's own update announcement, which is how a tool that already reports its own updates is read rather than reimplemented. +A tool does not always announce a new release on the command that prints its version: `no-mistakes --version` prints only the version, while its other commands carry the announcement. +`announce_args` names the command to search for the announcement in that case, and it is asked only of the copy `PATH` resolves; without it the version probe's own output is searched. +An `announce_pattern` that is not a usable extended regular expression stops `arm`, and during a sweep it is reported as that one tool's own check failure so one broken pattern never stops the other watched tools from being checked. +A `git` entry reports how many commits the local clone is behind its remote branch, and stays silent when the clone is current or ahead. +An omitted `branch` uses the remote's default branch, taken from the clone's own record of it and otherwise asked of the remote directly, so a `--single-branch` clone still resolves. Both probe kinds are read-only and bounded, and a probe that cannot answer is reported as a check failure rather than assumed current. See [`docs/examples/watched-tools.json`](examples/watched-tools.json) for a starting point to copy into local `config/watched-tools.json`. -**Arm, edit, and disarm** - Arm the check once per home with `bin/fm-tool-update-check.sh arm`. - -- That writes `state/tool-updates.check.sh` and binds its bytes with `bin/fm-check-register.sh`, so the existing watcher polls it on its normal cadence and turns its one line into a `check:` wake; no separate schedule is involved. -- Registering the check is itself a reason to watch, so the home keeps a watcher for it after the last task is torn down, and `disarm` is what ends that need. -- `bin/fm-tool-update-check.sh disarm` removes the shim, its trust binding, and the report record. - -**Repeat reporting and inheritance** - -- The check prints nothing when everything is current, and `state/.tool-updates` records the findings the last report was made from so the same pending update is reported once instead of on every poll. -- A changed or returning condition is reported again. -- Adding, removing, or changing a watched tool is an edit to this file and needs no code change or re-arming. -- This file is not inherited by secondmate homes, so each home watches the tools it actually depends on. - -**Probe timing and limits** - -| Setting | Default | Purpose | -| --- | --- | --- | -| `FM_TOOL_UPDATE_INTERVAL` | 900 seconds | Time between probes; `0` probes on every run. | -| `FM_TOOL_UPDATE_PROBE_SECS` | 5 | Bounds one probe. | -| `FM_TOOL_UPDATE_BUDGET_SECS` | 20 | Bounds a whole sweep. | - -- A sweep that runs out of budget says which tool it did not reach rather than reporting the rest as current. -- The sweep must finish inside `FM_CHECK_TIMEOUT` (default 30), because a run the watcher kills prints nothing and records nothing and would then repeat that silence on every poll. -- So a budget larger than that timeout allows is cut down to what fits instead of being refused, and the cut is reported in the report line. -- A budget that is not a whole number from 1 to 120 is still refused outright. +That writes `state/tool-updates.check.sh` and binds its bytes with `bin/fm-check-register.sh`, so the existing watcher polls it on its normal cadence and turns its one line into a `check:` wake; no separate schedule is involved. +Registering the check is itself a reason to watch, so the home keeps a watcher for it after the last task is torn down, and `disarm` is what ends that need. +`bin/fm-tool-update-check.sh disarm` removes the shim, its trust binding, and the report record. +The check prints nothing when everything is current, and `state/.tool-updates` records the findings the last report was made from so the same pending update is reported once instead of on every poll. +A changed or returning condition is reported again. +Adding, removing, or changing a watched tool is an edit to this file and needs no code change or re-arming. +This file is not inherited by secondmate homes, so each home watches the tools it actually depends on. + +`FM_TOOL_UPDATE_INTERVAL` (default 900 seconds, `0` to probe on every run) sets how often probes actually run, `FM_TOOL_UPDATE_PROBE_SECS` (default 5) bounds one probe, and `FM_TOOL_UPDATE_BUDGET_SECS` (default 20) bounds a whole sweep. +A sweep that runs out of budget says which tool it did not reach rather than reporting the rest as current. +The sweep must finish inside `FM_CHECK_TIMEOUT` (default 30), because a run the watcher kills prints nothing and records nothing and would then repeat that silence on every poll. +So a budget larger than that timeout allows is cut down to what fits instead of being refused, and the cut is reported in the report line. +A budget that is not a whole number from 1 to 120 is still refused outright. ## Mail plane (.env) The mail plane (bin/fm-mail.sh) reads unseen IMAP messages and sends one SMTP message. - -**Polling and delivery guarantees** - Its `poll` command surfaces each new message as a durable `check: mail ` wake, which is also what the standing received-mail check runs each watcher cycle. Poll emission is exactly-once-recovering: a published wake always carries a durable journal record, and a poll interrupted before recording its uid is healed from that journal, so inbound mail is never silently missed. - A duplicate wake is possible if the process is killed between the queue append and the journal write and the drain acknowledges that row before the next poll heals it, or under a triple write fault that leaves a queued row with no durable record; neither case drops mail. - -**Connection and activation** - IMAP and SMTP use implicit TLS on the default ports 993 and 465 (`IMAP4_SSL` / `SMTP_SSL`). STARTTLS and port 587 are not supported. - It is off unless the home's gitignored `.env` provides the connection values. This section is the single owner of the mail-plane configuration schema; for direct invocations, environment values override `.env`, matching the Relay contract. @@ -1381,20 +682,13 @@ FM_SMTP_HOST= # SMTP server hostname `FM_IMAP_PORT` (default 993), `FM_SMTP_PORT` (default 465), `FM_MAIL_TIMEOUT` (default 20 seconds), and `FM_MAIL_POLL_MAX_WAKES` (default 20, valid 1..200) are optional. The per-poll wake cap bounds the wakes of one `poll` run; header fetches scan a larger bounded window of new unseen uids plus already-surfaced retry-set uids, so a flood or large backlog still makes bounded progress every poll, keeping the durable wake queue bounded without ever dropping mail. - -**Unfetchable headers** - A message whose header cannot be fetched is surfaced with a degraded summary instead of being skipped, so it is never missed and cannot block later mail. A later poll retries that fetch and, on success, surfaces the real sender and subject; a persistently unfetchable message stays degraded without repeating that wake. -**Arm unattended polling** - A home that wants mail polled unattended arms the standing check in the live home: `bin/fm-mail-check.sh arm`. Arming writes `state/mail.check.sh` and registers it with the watcher's slow-check cadence (`FM_CHECK_INTERVAL`), so the plane's `poll` runs on its own: new mail still surfaces as `check: mail ` wakes from the poll, and the standing check itself also prints a line (and the watcher turns that line into a wake) unless the poll is a proven no-op. - Same-line silence is only for a proven no-op: a successful poll with no new mail, or a repeated identical pre-wake failure that cannot have queued mail. A fail-closed poll that already queued a wake, and a timeout, always print so the watcher wakes to drain it. - `FM_MAIL_CHECK_BUDGET` (default 15, valid 5..25) bounds one standing poll and is cut down to fit `FM_CHECK_TIMEOUT`. `bin/fm-mail-check.sh disarm` removes the standing check. @@ -1402,295 +696,135 @@ A fail-closed poll that already queued a wake, and a timeout, always print so th Relay lets a firstmate instance answer public mentions and act on normal reversible mention requests through firstmate's normal lifecycle. It covers both public surfaces the relay supports: `@myfirstmate` mentions on X, and mentions of the myfirstmate bot in a Discord server where it is installed. - Both surfaces are the same opt-in and the same machinery - one pairing token, one relay poll, and one reply path - so everything below applies to Discord mentions unless a line names a platform explicitly. - -**Activation, consent, and routing** - It is off unless the firstmate home's gitignored `.env` contains a non-empty `FMX_PAIRING_TOKEN`. The pairing token both identifies the relay tenant and records opt-in consent for autonomous public replies and eligible lifecycle actions. - Destructive, irreversible, or security-sensitive asks are flagged for trusted-channel confirmation instead of being executed from a public mention. The relay uses owner-only routing: a mention delivered to a home is from that home's owner/captain, while its surrounding conversation context may still include other public accounts. - -**Endpoint and environment overrides** - `FMX_RELAY_URL` is optional and defaults to `https://myfirstmate.io`, mainly for developers pointing at a local relay. For direct client invocations, environment values override `.env`; bootstrap activation still keys off `.env` presence so watcher artifacts are explicit local opt-in state. - `FMX_ENV_FILE` can point direct poll/reply client invocations at another `.env`-style file, but it does not change bootstrap activation. To turn it on: 1. Sign in at [myfirstmate.io](https://myfirstmate.io) with X or Discord. 2. For the Discord surface, use the dashboard's install link to add the myfirstmate bot to a server you administer; the X surface needs no install step. - 3. Copy the pairing token from the dashboard into this firstmate home's gitignored `.env` as `FMX_PAIRING_TOKEN=`. 4. Start a new firstmate session so bootstrap picks the token up, then mention `@myfirstmate` on X or mention the bot in a server where it is installed. The dashboard owns account creation, identity linking, bot installation, and token issuance; this document owns only what the local firstmate home does with the token once it is in `.env`. -**Generated state and watcher cadence** - The locked session-start bootstrap step turns the token into local generated state. It writes `state/x-watch.check.sh`, a byte-static identity shim for `bin/fm-x-poll.sh`, and `config/x-mode.env`, which exports `FM_CHECK_INTERVAL=30` for watcher processes in that home. - The watcher accepts the shim only when its bytes match the expected generated content, then invokes the trusted repository poll script directly instead of executing state-file source. -This section owns the Relay cadence contract: - -- A Relay instance polls every 30 seconds instead of the default 300. -- A non-Relay home has no `config/x-mode.env`, so its cadence does not change. -- When that file exists, the session-start supervision operating block includes the cadence instruction. - +This section is the single owner of the Relay cadence contract: a Relay instance polls every 30 seconds instead of the default 300, only a Relay instance speeds up because a non-Relay home has no `config/x-mode.env`, and the session-start supervision operating block includes the cadence instruction when that file exists. The active primary-harness supervision protocol owns how that sourced cadence reaches the watcher process. - -**Apply cadence changes** - Because `bin/fm-watch.sh` reads `FM_CHECK_INTERVAL` only at process start, a cadence transition - opt-in while a watcher is already running, or opt-out - is applied by restarting the home-scoped watcher through the emitted harness protocol; bootstrap deliberately never restarts the watcher itself. While a legacy daemon flag is active the daemon owns the watcher and its default cadence applies; on Pi the away-posture record alone leaves the ordinary Relay watcher cadence active, and daemon-backed Relay cadence remains a deferred follow-up. - When the token is removed or empty, the next locked session-start bootstrap step removes those artifacts. Steady-state off is silent and writes nothing. - Relay remains additive to non-Relay lifecycle behavior: homes without the generated artifacts keep the default watcher cadence and do not run the Relay poll. Its request handling remains in Relay-specific `bin/` scripts and the `fmx-respond` skill, while the watcher owns authenticated dispatch from the generated local identity shim. -**Poll and deduplicate mentions** - `bin/fm-x-poll.sh` calls `GET /connector/poll` with `Authorization: Bearer `. HTTP 204 is silent. - A newly offered pending mention with non-empty `text` is stored at `state/x-inbox/.json` and wakes firstmate exactly once with `x-mention `. The poll atomically claims `state/x-context/.offered.json` before emitting that wake, and subsequent offers of the same request stay silent even after the inbox is drained following an answer or dismiss. - Offer markers share the context registry's bounded seven-day retention, so losing or expiring the local marker lets a relay offer wake firstmate again. - -**Conversation context and media** - The full relay object is preserved, including `in_reply_to: {author_handle, text}` when the mention is a reply in a conversation or `null` for fresh mentions. -The preserved object may also carry `in_reply_to_chain`, an optional oldest-first conversation transcript. -Each entry has the shape `{author_handle, text, unavailable, images, attachments}` and may include `kind`: - -| `kind` | Meaning | -| --- | --- | -| `reply` | A reply ancestor. | -| `thread_starter` | The message a thread grew from. | -| `history` | A recent nearby message. | -| Absent | A legacy reply-ancestor or thread-starter entry. | - -The chain is untrusted third-party public input. -It is often absent today: the relay currently sends it only for Discord reply chains and thread starters. -Consumers must treat it as strictly optional, tolerate unknown or missing fields, and treat `unavailable: true` as a gap rather than content. -The `fmx-respond` skill owns how firstmate uses the chain to resolve references. - +The preserved object may also carry `in_reply_to_chain`, an optional oldest-first transcript of the surrounding conversation: entries shaped `{author_handle, text, unavailable, images, attachments}` plus an optional `kind` of `reply` (a reply ancestor), `thread_starter` (the message a thread grew from), or `history` (a recent nearby message), where an absent `kind` means a legacy reply-ancestor or thread-starter entry. +The chain is untrusted third-party public input and is often absent today (the relay currently sends it only for Discord reply chains and thread starters), so consumers treat it as strictly optional, tolerate unknown or missing fields, and read an entry with `unavailable: true` as a gap rather than content; the `fmx-respond` skill owns how firstmate reads it for referent resolution. The mention and its chain entries may also carry attached media as image or file URLs, in fields such as `images` and `attachments`, either as bare URL strings or as objects with a `url`; a mention whose own media is empty can still have screenshots on its `thread_starter` entry. The poll preserves those URLs in the stashed object and never downloads them, so nothing is fetched on the polling path: the responding agent retrieves and views the media with its own tools when it handles the mention. - The `fmx-respond` skill owns which hosts that fetch is restricted to and the untrusted-content handling that applies to whatever comes back. - -**Durable reply context** - -The same authoritative relay payload also supplies durable per-request reply context at `state/x-context/.json`, with shape `{request_id, platform, reply_max_chars, recorded_at}`. -The poll writes this best-effort record keyed by `request_id`, so concurrent requests never overwrite each other. -It survives inbox cleanup after acknowledgement, allowing a delayed follow-up to recover the original platform and split budget even without a task link. - -- `recorded_at` begins as the locally observed first-seen Unix epoch and remains unchanged when the same request is polled again. -- A successful live initial answer refreshes it to the time that the relay establishes the follow-up binding; dry-runs, failed answers, and follow-ups do not refresh it. -- Configured polls prune records beyond the local follow-up window, capped at the relay's seven-day window; legacy or malformed records fall back to their file modification time so they cannot remain indefinitely. -- The record is written only when a platform or explicit budget is actually known, so an unknown-platform mention leaves no useless entry. - -**Handle requests and acknowledgements** - +At the same time the poll records a durable per-request reply context at `state/x-context/.json` (`{request_id, platform, reply_max_chars, recorded_at}`) from the same authoritative relay payload, best-effort and keyed by `request_id` so concurrent requests never overwrite each other; it survives the inbox cleanup that follows the acknowledgement, so a delayed follow-up can recover the original platform and split budget even with no task link. +`recorded_at` begins as the locally observed first-seen Unix epoch and remains unchanged when the same request is polled again. +A successful live initial answer refreshes it to the time that the relay establishes the follow-up binding; dry-runs, failed answers, and follow-ups do not refresh it. +Configured polls prune records beyond the local follow-up window, capped at the relay's seven-day window; legacy or malformed records fall back to their file modification time so they cannot remain indefinitely. +The record is written only when a platform or explicit budget is actually known, so an unknown-platform mention leaves no useless entry. The `fmx-respond` skill decides whether the stashed mention is an actionable request, a question, or a pure acknowledgment. - -- Actionable reversible requests are run through intake, backlog, dispatch, investigation, or ship flow as appropriate. -- If the work completes in that turn, the public reply reports the outcome. -- If the request spawns a longer-running task, firstmate posts an acknowledgement through the normal answer endpoint, links the task to the mention with `bin/fm-x-link.sh`, and posts up to three completion follow-ups on genuine milestones, finishing with a `--final` one for ordinary Relay-linked work. - When a typed promised-final commitment is registered, `bin/fm-public-followup.sh` owns the terminal reply and clears the legacy link after its receipt is validated. -- That link stores optional reply-platform context so Discord-originated follow-ups keep Discord's larger message budget after the inbox file has been drained. - -**Resolve the reply platform and budget** - +Actionable reversible requests are run through intake, backlog, dispatch, investigation, or ship flow as appropriate. +If the work completes in that turn, the public reply reports the outcome. +If the request spawns a longer-running task, firstmate posts an acknowledgement through the normal answer endpoint, links the task to the mention with `bin/fm-x-link.sh`, and posts up to three completion follow-ups on genuine milestones, finishing with a `--final` one for ordinary Relay-linked work. When a typed promised-final commitment is registered, `bin/fm-public-followup.sh` owns the terminal reply and clears the legacy link after its receipt is validated. +That link stores optional reply-platform context so Discord-originated follow-ups keep Discord's larger message budget after the inbox file has been drained. Platform/budget resolution is layered and independent of the task link: a per-axis `FMX_REPLY_PLATFORM` / `FMX_REPLY_MAX_CHARS` override (how `bin/fm-x-followup.sh` passes a recorded link's context) wins. -For either axis without an override, `bin/fm-x-lib.sh:fmx_resolve_reply_context` consults these sources in order: - -1. The durable per-request registry. -2. The still-present inbox payload. -3. For a follow-up posted live by request_id only, an authoritative relay lookup through `POST /connector/request-context`: `{request_id}` in, `{platform, reply_max_chars}` back. - +For either axis without an override, `bin/fm-x-lib.sh:fmx_resolve_reply_context` owns the source order: the durable per-request registry is consulted first, then the still-present inbox payload, then - for a follow-up posted live by request_id - an authoritative relay lookup via `POST /connector/request-context` (`{request_id}` in, `{platform, reply_max_chars}` back). This is what keeps a delayed request-id follow-up on the original platform's budget even after the inbox is drained and with no task link surviving; the relay step is confined to the live follow-up path so the answer path and every dry-run stay network-free. - -**Link tasks and handle missing context** - -The link lives in the current home's `state/.meta`. -Work routed to a secondmate has no record here, so `bin/fm-x-link.sh` refuses to link it. -When possible, the refusal names the registered secondmate home containing the task. - -It also points to `bin/fm-public-followup.sh register ... --work-home secondmate:`. -This promised-final path is the only follow-up mechanism that binds work in another home. -`bin/fm-x-link.sh` uses the same order when recording a fresh link's context and requires `jq`. -Its request-context lookup is best-effort. -Any of these conditions leaves the context unknown: - -- No token or `curl`. -- A non-2xx response. -- An unresolved response. -- A relay version without that endpoint. - -The link is still recorded, but `bin/fm-x-link.sh` prints a loud warning. -If either the follow-up platform or explicit budget cannot be authoritatively resolved from any source, `bin/fm-x-reply.sh` refuses with fail-safe exit 8. -Firstmate holds the follow-up and retries once both values are recoverable; it never posts with a local default. - -**Carry a link to a successor task** - +The link is home-local by construction, because it lives in that home's own `state/.meta`: work routed to a secondmate has no record here, so `bin/fm-x-link.sh` refuses it, names the registered secondmate home the task was found in when it can, and points at the promised-final path (`bin/fm-public-followup.sh register ... --work-home secondmate:`), which is the only follow-up mechanism that binds work in another home. +`bin/fm-x-link.sh` follows the same ordering when recording a fresh link's context and requires `jq`; its request-context lookup is best-effort: no token or `curl`; a non-2xx response; an unresolved response; or a relay version without that endpoint leaves the context unknown. +In that case the link is still recorded but `bin/fm-x-link.sh` prints a loud warning; and when either a follow-up's platform or explicit budget cannot be authoritatively resolved from any source, `bin/fm-x-reply.sh` refuses it (fail-safe exit 8) rather than posting with a local default - firstmate holds and retries it once both values are recoverable. Fresh links start with `x_followups=0` and the current timestamp; when relinking the same relay request onto a successor task, pass paired `--carry-count --carry-ts ` flags plus any prior `x_platform=` and `x_reply_max_chars=` as `--carry-platform --carry-max ` so the successor preserves the already-consumed follow-up count, original 7-day window, and reply split budget. - -**Dismiss mentions** - Pure acknowledgments or mentions with nothing to answer are dismissed through `bin/fm-x-dismiss.sh` before the local inbox file is cleared. Dismiss sends `POST /connector/dismiss` with `{request_id}`, posts no text, and tells the relay to drop the request instead of re-offering it or falling back to an offline auto-reply; on success it clears that request's durable reply-context record, while the separate offer marker remains for its bounded retention so a brief relay re-offer stays silent. - -**Poll errors** - Relay auth or config problems are reported once as `x-mode-error ...` until recovery. A failed durable offer claim is likewise reported once as `x-mode-error cannot record mention offer` and remains deduplicated through quiet no-pending polls until a later offer confirms an existing valid marker or claims a new one. - -**Post replies and follow-ups** - Live replies are posted by `bin/fm-x-reply.sh`, which sends `POST /connector/answer` with `{request_id,text}` for one-message replies. Add `--image ` to attach one local PNG, JPEG, GIF, WebP, BMP, or TIFF as `{media_type,data_base64}` in the relay's optional `image` object. - Completion follow-ups use `bin/fm-x-followup.sh`, which checks the local `state/.meta` link and sends the same payload shape through `POST /connector/followup` by calling `bin/fm-x-reply.sh --followup`, up to three times per link within the window. Add `--image ` there too when a completion follow-up should carry an image. - -**Follow-up success, expiry, and retry** - -- A successful post increments the local `x_followups=` counter and keeps the link, unless `--final` was passed or the new count reaches the cap, in which case the link is cleared instead; a failed post leaves the link and counter untouched so it can be retried. -- The relay itself rejects a follow-up past its own cap or window with HTTP 409 and may include `{"error":"followup_unavailable"}` in the response body; the client surfaces any follow-up 409 as a distinguishable exit code and uses the body marker only for a sharper diagnostic. -- `fm-x-followup.sh` treats that exit exactly like a locally-detected expiry - clearing the link and skipping quietly rather than retrying - so an older single-follow-up relay or an already-exhausted binding degrades gracefully. -- It treats `fm-x-reply.sh`'s fail-safe refusal (exit 8: platform or explicit budget unresolved) differently: that is a retryable hold, so the link is KEPT and the follow-up is retried once both values can be recovered, never posted with a local default. -- Past-window relay rejections are only guaranteed while the expired binding row still exists on the relay side; after its cleanup sweep, a very-late follow-up call may instead see a benign no-op 200, which is why the local window and cap pruning remains the primary guard. - -**Split replies by platform** - -- Reply splitting is platform-aware: an explicit relay platform field (`reply_platform`, `platform`, `target_platform`, `source_platform`, or `provider`) wins, otherwise a legacy `tweet_id` beginning with `discord:` selects Discord and a numeric `tweet_id` selects X. -- An explicit relay limit field (`reply_max_chars`, `reply_max_characters`, `message_max_chars`, `message_limit`, or `max_chars`) wins over the platform defaults. -- If the reply exceeds the selected budget, the client splits it into a numbered thread on fenced-code, paragraph, line, and word boundaries and sends `{request_id,text,texts}`, where `texts` is the ordered chunk list and `text` remains the first chunk for older relays. -- When `--image ` is present on a split reply, the image rides the first/opener message and later chunks stay text-only. - -**Reply and follow-up limits** - -| Setting | Default | Limit or behavior | -| --- | --- | --- | -| `FMX_X_REPLY_MAX_CHARS` | 280 | Clamps to a minimum of 50. | -| `FMX_DISCORD_REPLY_MAX_CHARS` | 1900 | Clamps to a minimum of 50; values above Discord's 2000-character limit reset to 1900. | -| `FMX_X_THREAD_MAX` | 25 | Caps oversized reply threads on every platform; truncation marks the last retained message with an ellipsis. | -| `FMX_FOLLOWUP_MAX_AGE_SECS` | 604800 (7 days) | Local completion follow-up window. | -| `FMX_FOLLOWUP_MAX_COUNT` | 3 | Local follow-up cap. | - -**Preview with dry-run** +A successful post increments the local `x_followups=` counter and keeps the link, unless `--final` was passed or the new count reaches the cap, in which case the link is cleared instead; a failed post leaves the link and counter untouched so it can be retried. +The relay itself rejects a follow-up past its own cap or window with HTTP 409 and may include `{"error":"followup_unavailable"}` in the response body; the client surfaces any follow-up 409 as a distinguishable exit code and uses the body marker only for a sharper diagnostic. +`fm-x-followup.sh` treats that exit exactly like a locally-detected expiry - clearing the link and skipping quietly rather than retrying - so an older single-follow-up relay or an already-exhausted binding degrades gracefully. +It treats `fm-x-reply.sh`'s fail-safe refusal (exit 8: platform or explicit budget unresolved) differently: that is a retryable hold, so the link is KEPT and the follow-up is retried once both values can be recovered, never posted with a local default. +Past-window relay rejections are only guaranteed while the expired binding row still exists on the relay side; after its cleanup sweep, a very-late follow-up call may instead see a benign no-op 200, which is why the local window and cap pruning remains the primary guard. +Reply splitting is platform-aware: an explicit relay platform field (`reply_platform`, `platform`, `target_platform`, `source_platform`, or `provider`) wins, otherwise a legacy `tweet_id` beginning with `discord:` selects Discord and a numeric `tweet_id` selects X. +An explicit relay limit field (`reply_max_chars`, `reply_max_characters`, `message_max_chars`, `message_limit`, or `max_chars`) wins over the platform defaults. +If the reply exceeds the selected budget, the client splits it into a numbered thread on fenced-code, paragraph, line, and word boundaries and sends `{request_id,text,texts}`, where `texts` is the ordered chunk list and `text` remains the first chunk for older relays. +When `--image ` is present on a split reply, the image rides the first/opener message and later chunks stay text-only. +`FMX_X_REPLY_MAX_CHARS` defaults to 280 and clamps to a minimum of 50; `FMX_DISCORD_REPLY_MAX_CHARS` defaults to 1900, clamps to a minimum of 50, and resets values above Discord's 2000-character limit back to 1900. +`FMX_X_THREAD_MAX` defaults to 25 and caps oversized reply threads for every platform, marking the last retained message with an ellipsis when truncation is needed. +`FMX_FOLLOWUP_MAX_AGE_SECS` defaults to 604800 (7 days) and controls the local completion follow-up window; `FMX_FOLLOWUP_MAX_COUNT` defaults to 3 and controls the local follow-up cap. Set `FMX_DRY_RUN` to preview replies and dismissals without posting. Truthy means anything except unset, empty, `0`, `false`, `no`, or `off`; an explicit environment value wins over `.env`. - -- In dry-run, `fm-x-reply.sh` records the would-be payload to `state/x-outbox/.json`, including `texts` for a thread and an `endpoint` marker for follow-up previews, prints a `DRY RUN` summary to stderr, echoes the `request_id`, and exits 0. -- When an image is attached, the dry-run record uses compact `{media_type, bytes, source_path}` metadata instead of writing the base64 bytes. -- In dry-run, `fm-x-dismiss.sh` records `{request_id, endpoint:"dismiss"}` to the same outbox path, prints a `DRY RUN` summary, echoes the `request_id`, and exits 0. -- The live answer and follow-up bodies intentionally stay the same shape, including optional `image`; the relay distinguishes them by endpoint, and dismiss stays `{request_id}`. -- These paths need `jq` to build the JSON payload, but they run before token and network checks, so they need neither `FMX_PAIRING_TOKEN` nor `curl`. +In dry-run, `fm-x-reply.sh` records the would-be payload to `state/x-outbox/.json`, including `texts` for a thread and an `endpoint` marker for follow-up previews, prints a `DRY RUN` summary to stderr, echoes the `request_id`, and exits 0. +When an image is attached, the dry-run record uses compact `{media_type, bytes, source_path}` metadata instead of writing the base64 bytes. +In dry-run, `fm-x-dismiss.sh` records `{request_id, endpoint:"dismiss"}` to the same outbox path, prints a `DRY RUN` summary, echoes the `request_id`, and exits 0. +The live answer and follow-up bodies intentionally stay the same shape, including optional `image`; the relay distinguishes them by endpoint, and dismiss stays `{request_id}`. +These paths need `jq` to build the JSON payload, but they run before token and network checks, so they need neither `FMX_PAIRING_TOKEN` nor `curl`. ### Promised public replies (state/public-followup) A relay request that spawns real work can leave firstmate owing a specific public reply in a specific thread. That promise is a typed `kind=public-followup` obligation whose state machine is owned entirely by `tasks-axi public-followup`, while the full private conversation context stays only in `state/x-context/`. - Firstmate's bounded registration retains the obligation's public-safe request binding so a delivered loop can be rechained without the original inbox. `bin/fm-public-followup.sh` is firstmate's side: it registers a commitment, reconciles typed terminal work results into it, posts the final reply through `bin/fm-x-reply.sh --followup`, and explicitly rechains or retires the retained loop. - Run `bin/fm-public-followup.sh --help` for the exact subcommands and flags. -**Private transport records** - -Registration creates this home's private transport under `state/public-followup/` with mode 0700: - -| Entry | Purpose and retention | -| --- | --- | -| `registry/` | Bounded private binding for each open public loop; survives delivery with `state=delivered`; only `retire` removes it. | -| `events/` | Typed terminal results awaiting reconciliation. | -| `consumed/` | Accepted-event ledger. | -| `rejected/` | Refusals retained with a one-line reason. | -| `rejection-wakes/` | Each refusal's not-yet-raised wake. | -| `retired/` | Mode-0600 reason-and-time receipt written before removal. | -| `surfaced` | The poll's last-surfaced signature. | -| `outbox/` | Also created in a work home that reports across a machine boundary; described below. | - -**Which home posts the reply** - +Registration is what creates this home's private transport under `state/public-followup/` (mode 0700): `registry/` for the bounded private binding of each open public loop (the record survives delivery, stamped `state=delivered`, and is removed only by `retire`), `events/` for typed terminal results awaiting reconciliation, `consumed/` for the accepted-event ledger, `rejected/` for refusals kept with a one-line reason, `rejection-wakes/` for each refusal's not-yet-raised wake, `retired/` for the mode-0600 reason-and-time receipt written before removal, and `surfaced` for the poll's last-surfaced signature. +A work home that reports across a machine boundary also gets `outbox/`, described below. The home that owns the commitment also owns the outward post, because only it holds the relay consent, the request context, and the opaque thread binding. Work routed elsewhere reports a typed terminal result with `bin/fm-public-followup-emit.sh` and never looks for the thread; when writing directly into the owning home, that emitter refuses a home with no registration for the named obligation. - -**Prepare and validate terminal results** - `bin/fm-public-followup.sh brief` pre-fills every deliverable value the binding determines, such as `report_path=data//report.md`, and states the accepted format of every value it cannot know. - -- The emitter validates deliverable values and known required keys before publishing, including the relative `report_path` format, and names correctable mistakes at the work home. -- A direct emit reads the obligation from `tasks-axi`; a staged emit cannot read that remote record, so `brief` supplies its required keys in the printed command. -- If those flags are omitted from a staged command, it still checks values but cannot detect missing keys until the owning home's `consume` rejects the event and queues a rejection wake. -- The [emitter header](../bin/fm-public-followup-emit.sh) and its `--help` own the exact flags and outcome-dependent validation rules. - -**Clear legacy links in remote homes** - +The emitter validates deliverable values and known required keys before publishing, including the relative `report_path` format, and names correctable mistakes at the work home. +A direct emit reads the obligation from `tasks-axi`; a staged emit cannot read that remote record, so `brief` supplies its required keys in the printed command. +If those flags are omitted from a staged command, it still checks values but cannot detect missing keys until the owning home's `consume` rejects the event and queues a rejection wake. +The [emitter header](../bin/fm-public-followup-emit.sh) and its `--help` own the exact flags and outcome-dependent validation rules. When that work lives in a REMOTE secondmate home, delivery clears its bound legacy link after validating the public receipt, while retirement clears the link before closing the loop, and both clears run over that route's SSH transport. -Readable remote state proving that no link exists succeeds without a write. -A present link is cleared only when its Relay request identity matches the registration and the state is writable. -Any of these conditions retains the loop for reconciliation: - -- An identity mismatch. -- Unreadable or unsafe state. -- An unavailable write or lock. -- An older remote copy. -- A host that never confirms the clear. - -**Duplicate and failed results** - +Readable remote state that proves no link exists succeeds without a write, while a present link is cleared only when its Relay request identity matches the registration and the state is writable; an identity mismatch, unreadable or unsafe state, an unavailable write or lock, an older remote copy, or a host that never confirms the clear leaves the loop retained for reconciliation. A terminal event's id is derived from its identity tuple, so a duplicate report, a retry, or a replay after restart resolves to the same event and changes nothing. When bound work ends failed or parked, its typed failed result remains deliverable even when the promised final expected a merged pull request, so the owed reply carries the honest failure instead of remaining stranded. -**Collect results across machines** - Work bound to a REMOTE secondmate home reports across a machine boundary, where no local path reaches the owning home. - -- `bin/fm-public-followup.sh brief` therefore prints that worker the route's own code root and home with `--stage-in`, so the typed result is staged in `outbox/` in the home where the work actually runs rather than written to a path that only exists on the owning machine. -- The owning home collects staged results for open registrations over the same SSH route it reaches that secondmate on, because that transport only runs in the outbound direction: `consume` pulls them into its own `events/` and then reconciles them exactly as it reconciles a local report. -- Non-open registrations owe no result, so `consume` skips them without contacting their routes; an open registration whose reachable route has nothing staged remains pending without an error. -- Collection is non-destructive until the result is durably held, and the staged copy is retired only afterwards, so a dropped connection can never lose a terminal result. -- For an open registration, a work home that cannot be reached is named in `consume`'s output and keeps the promise open; it is never reported as an empty inbox. - +`bin/fm-public-followup.sh brief` therefore prints that worker the route's own code root and home with `--stage-in`, so the typed result is staged in `outbox/` in the home where the work actually runs rather than written to a path that only exists on the owning machine. +The owning home collects staged results for open registrations over the same SSH route it reaches that secondmate on, because that transport only runs in the outbound direction: `consume` pulls them into its own `events/` and then reconciles them exactly as it reconciles a local report. +Non-open registrations owe no result, so `consume` skips them without contacting their routes; an open registration whose reachable route has nothing staged remains pending without an error. +Collection is non-destructive until the result is durably held, and the staged copy is retired only afterwards, so a dropped connection can never lose a terminal result. +For an open registration, a work home that cannot be reached is named in `consume`'s output and keeps the promise open; it is never reported as an empty inbox. Run `bin/fm-public-followup-collect.sh --help` for the staged-result commands the owning home runs over that route. -**Activation and idle cost** - Activation is the same `.env` `FMX_PAIRING_TOKEN` contract as the rest of Relay, with no second flag. - -- A home without that token runs one file test and stops: no `tasks-axi` call, no backlog or request-context scan, and no `state/public-followup/` directory. -- Ordinary startup, polling, cleanup, and silent read-side subcommands also produce no output; commands that require an active relay report that configuration error after the same gate. -- A relay-enabled home with no registered commitment stops at an O(1) directory presence check, so the empty state costs no CLI call and adds no periodic scan. - -**Wake on new or rejected results** - +A home without that token runs one file test and stops: no `tasks-axi` call, no backlog or request-context scan, and no `state/public-followup/` directory. +Ordinary startup, polling, cleanup, and silent read-side subcommands also produce no output; commands that require an active relay report that configuration error after the same gate. +A relay-enabled home with no registered commitment stops at an O(1) directory presence check, so the empty state costs no CLI call and adds no periodic scan. Unreconciled terminal results ride the existing 30-second relay poll rather than a new process or timer: `bin/fm-x-poll.sh` compares the pending-event signature against `surfaced` and wakes firstmate once per new result set. - -- A terminal event `tasks-axi` refuses during `consume` is quarantined with a reason naming the specific deliverable, outcome, or missing key where one is identifiable, and the same poll wakes the owning home with a `public-followup rejected ...` line carrying that reason. -- The refused event stays pending until that wake is recorded, and a queued wake survives a failed read or write to poll output. -- That makes the wake at-least-once rather than exactly-once: a cleanup that fails after the line was already raised - a wake directory that cannot be written, or a refused event that could not be drained - raises the same refusal again on a later poll. -- A repeat carries the same event id and the same reason as the quarantined rejection, which is how an already-handled refusal is recognized. -- Acknowledge it without re-acting; re-emitting an already accepted corrected result is harmless but redundant because its derived event id is already in the accepted ledger. - -**Startup, teardown, and retries** - +A terminal event `tasks-axi` refuses during `consume` is quarantined with a reason naming the specific deliverable, outcome, or missing key where one is identifiable, and the same poll wakes the owning home with a `public-followup rejected ...` line carrying that reason. +The refused event stays pending until that wake is recorded, and a queued wake survives a failed read or write to poll output. +That makes the wake at-least-once rather than exactly-once: a cleanup that fails after the line was already raised - a wake directory that cannot be written, or a refused event that could not be drained - raises the same refusal again on a later poll. +A repeat carries the same event id and the same reason as the quarantined rejection, which is how an already-handled refusal is recognized. +Acknowledge it without re-acting; re-emitting an already accepted corrected result is harmless but redundant because its derived event id is already in the accepted ledger. The session-start digest separately prints a "Public commitments" subsection from disk when, and only when, this home is relay-active and still holds an open public loop (a reply still owed, or a delivered loop with nothing owed), so compaction and restart are non-events. `bin/fm-teardown.sh` refuses to clean up a task while this home still owes a public reply for exactly that work, unless `--force` carries explicit discard approval. - `FM_PF_RETRY_BACKOFF_SECS` (default 900) sets the next-attempt time recorded with a retryable delivery error. See [verification/public-followup.md](verification/public-followup.md) for the current maintainer evidence behind restart recovery, failed terminal outcomes, retained-loop disposition, and the relay-disabled zero-overhead guarantee. @@ -1698,34 +832,18 @@ See [verification/public-followup.md](verification/public-followup.md) for the c A home can explicitly enable a trusted external `process-event-adapter/1` package without adding package code to Firstmate. This is one narrow extension type, not a general plugin or hook system. - [`extension-bindings.md`](extension-bindings.md) owns the manifest, binding, trust, handshake, invocation-envelope, capability, version-compatibility, and authority-boundary contracts. `bin/fm-extension.sh --help` and `bin/fm-procevent.sh --help` own exact command mechanics. -**Discovery and disabled behavior** - Discovery reads only mode-`0600` bindings under this home's mode-`0700` `config/extensions.d/` directory. The current directory, projects, task copies, worker text, environment payloads, and Pi packages are never searched for extensions. - When the directory is absent, ordinary process-event commands perform only a bounded absence check, create no package or extension state, and preserve every built-in adapter path. -**Bind a trusted package** - Binding separates the package's own manifest from this home's explicit enablement. -`bind` performs these steps: - -1. Validate the source package and compute every digest. -2. Copy the complete tree into the read-only content-addressed `data/extensions/packages/` store. -3. Perform the live handshake. -4. Atomically publish the enabled adapter-name subset. - +`bind` validates the source package, computes every digest, copies the complete tree into the read-only content-addressed `data/extensions/packages/` store, performs the live handshake, and atomically publishes the enabled adapter-name subset. The operator supplies trust and required consent facts, not hashes. - -**Working state and cleanup** - `state/extensions//` is created when binding performs its initial handshake and is that package's home-local working namespace for later verification and invocation. `state/extension-invocations/` contains private host-owned exact process-group cleanup records only while an enabled package invocation is starting or running; retirement and reconciliation retain their existing owners until those records prove the group extinct. - This integrity boundary does not sandbox trusted same-user code, so bind only a package trusted to run with the operator's operating-system access. The shipped `file-signal` package is a complete neutral example. @@ -1747,9 +865,6 @@ bin/fm-extension.sh verify org.firstmate.example.file-signal Use an absent destination for the copy so the source identity remains inspectable and reproducible. For a non-default home, set `FM_HOME=` on every command; local and remote secondmate homes bind the package independently, and bindings are not inherited. - -**Bind on a remote secondmate** - For a configured remote secondmate, keep the package at the controller and transfer it through the authenticated `fm-on` route: ```sh @@ -1762,16 +877,10 @@ bin/fm-extension.sh remote-bind \ The command serializes only the validated extension package, stages it below the addressed remote home's fixed extension staging root, binds it there, and prints transfer and binding digests. Registration uses `bin/fm-on.sh fm-procevent.sh ...`. - -**Retire a binding** - After retiring every registration with its printed owner token and handling every captured result, retire the enabled remote binding and its exact staged transfer together with `bin/fm-on.sh fm-extension.sh retire-transfer --if-transfer-digest --if-binding-digest `. For a direct local binding, use `bin/fm-extension.sh retire-binding --if-binding-digest ` after the same process-event retirement and handling steps. - Both commands retain the retired identity reversibly and leave unrelated bindings and content-addressed installed packages unchanged. -**Register a completion source** - Register one file completion source with a path-safe source id and an explicit non-secret source configuration reference. Credential values never belong in that reference, command argv, or a process-event result: @@ -1782,393 +891,191 @@ bin/fm-procevent.sh reconcile ``` `register-extension` prints the new registration's owner token and exact owner-matched retirement command. - -**Classify and acknowledge results** - The source waits outside the conversational turn, and its completed result arrives through the existing process-event `check` path. Classify the captured result through its immutable package identity with `bin/fm-procevent.sh classify `, acknowledge it with the existing `handled` command only after it is handled, and use the printed `retire --if-owner` command when explicit retirement is needed. - -**Keep blocking sources out of the turn** - Never run the registered blocking source command directly in a conversational turn. ## Process-to-event sources (state/procevent) A long-polling external process is registered as a *source* through its adapter, whose header and `--help` own the commands and flags. `bin/fm-procevent.sh` owns the generic contract; built-in adapters retain their tracked `bin/fm-procevent-.sh` commands, while an explicitly bound external adapter routes through the trusted host contract above. - `bin/fm-procevent-lavish.sh` is the first built-in adapter and wraps only the currently published `lavish-axi poll` interface. - -**Open the Lavish artifact first** - Before arming any Lavish source, open its artifact with `lavish-axi` so the saved session identifies the board's server; each poll attempt derives its host and port from that session and refuses missing or invalid session evidence before consuming a staged worker reply. - -**Retry interrupted Lavish polls** - That adapter, and only that adapter, retries the one exact transient response a cut-short listener returns while its marks remain available (`error: Lavish Editor poll response was interrupted` with `code: SERVER_ERROR`), up to 12 times with poll starts at least 5 seconds apart, so an internal retry never reaches the runner as a captured result. This start-to-start governor is a no-op after a normally blocking poll but caps an immediately returning poll under the shipped defaults independently of the owner lease and registration launch pacing. - Real feedback, ended and missing sessions, any other `SERVER_ERROR`, and that same interruption still standing once the bound is spent are all captured and announced normally; `FM_LAVISH_POLL_RETRY_DELAY` is a bounded 1 to 60 second test override for the interval only, and the runner itself stays adapter-agnostic. -An already-armed Lavish source keeps its registered listener command until it is retired and armed again, so retire the source, then arm it again to adopt this retry policy. +An already-armed Lavish source keeps its registered listener command until it is retired and armed again, so re-arm a live board once to adopt this retry policy. ### Crew-hosted Lavish review boards -**Arm and confirm a listener** - A live task that hosts a Lavish board owns its listener, so firstmate must never arm that board. After opening the artifact as required above, the worker arms it with `bin/fm-procevent-lavish.sh arm --for ` and never runs `lavish-axi poll` itself. - -`arm` prints `armed` only after the process-event owner confirms this registration generation's listener is running, and otherwise returns nonzero without that line. - -- The confirmation is the same live claim or launch-stamp evidence `reconcile` already uses, bounded by `FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS`, and a failed confirmation retires a source that never started unless `retire` refuses because something may still own it, in which case the registration stays for `reconcile` or a human. -- An earlier registration's listener that releases the board inside the confirm window lets the new registration start, and `arm` then reports `armed` as usual. -- When a live listener from an earlier registration of the same board still holds it when the window ends, `arm` exits zero with `still-listening` instead of `armed`, because that earlier listener keeps serving the board and the new registration takes effect only after the source is retired and armed again. -- The arm is refused unless that task id has valid, identity-matching endpoint metadata, because a board whose owner has no endpoint would collect feedback nobody can be told about. - -**Acknowledge a round by re-arming** - +The arm is refused unless that task id has valid, identity-matching endpoint metadata, because a board whose owner has no endpoint would collect feedback nobody can be told about. The registration persists as one task-owned source record, while each captured nonterminal round remains open until the worker re-arms and the existing handled marker acknowledges that round. Re-arm is that acknowledgement and nothing else: the board is armed once while no record exists, and a further arm by the same owner is refused unless an unacknowledged nonterminal round is waiting, so a generation already carrying a reply is never replaced before its listener posts it. - -**Stage an agent reply** - -Re-arm never acquires, releases, or hands off the source claim. -It may carry `--agent-reply-file `. -The file's contents are copied into that generation's private staging file and passed once to the published `--agent-reply` argument. - -A failed re-arm leaves the prior registration and its referenced reply unchanged, including when its required acknowledgement cannot be recorded. -Reply posting is best effort by design. -The listener consumes the staged file only after validating its own setup and the board artifact. -The one loss window is a rare crash between consuming the file and making the call, which drops that round's reply rather than posting it twice. - -This path keeps no receipt, retry, or idempotency record. -Robust reply delivery waits on lavish-axi's exclusive listener. - -**Deliver feedback to the worker** - -- The captured result is stored with immutable task-owner routing evidence and delivered directly to that task's steering inbox, without a firstmate `check` wake for the captain's words. -- Filing that steering note away is not acknowledging the round, so while the round stays open every reconcile puts a live note back in the owner's inbox rather than ringing a filed one. -- A task-owned source with an unhandled capture is not relaunched, so delivery failure cannot consume a round and start another poll. -- That record is the only ownership evidence there is, so while any captured round of it is unacknowledged every retirement path refuses - the runner's own terminal retirement and an explicit `retire` alike - and the refusal names the acknowledgement that releases it. - -**Conclude a terminal round** - -- A terminal result, including `session_ended`, an empty End, or missing, is delivered to the owner with an explicit stop-and-conclude instruction and is never auto-rearmed. -- That round keeps the board with its owner: the source record is not retired while the terminal capture is unacknowledged, so no second armer can take the board, and acknowledging it with `bin/fm-procevent.sh handled ` is what concludes and retires it. -- That conclude retains the registration it is retiring, removes it, then records the acknowledgement and restores the registration if that record cannot be written, so a failed conclude never leaves the round open with its owner gone. -- An interruption between those two durable steps leaves the board unregistered with its terminal round still open, which nothing relaunches and the same `handled` call finishes. -- It concludes only a round that is still open, so a repeated acknowledgement of an already-closed round reports `already-handled` and never touches whatever registration holds the board by then. - -**Ownership and recovery** - -- A second armer is refused with the current owner named, and the source list derives `listening`, `round-open`, or `dead` from the claim and handled captures without a second ownership record. -- If the hosting worker cannot be recovered, relaunch a worker to re-host first; guarded firstmate adoption is an explicit last resort only after the old claim is proved dead. -- The cross-home gap between worker rounds remains an accepted residual until lavish-axi's exclusive listener lands. -- The interim crew instruction emitted by `bin/fm-brief.sh` points workers at this arm-and-acknowledge contract. - -**Register deterministic condition and action watches** - -The `when` adapter (`bin/fm-procevent-when.sh`) registers a deterministic condition and action once. -Its blocking child polls the condition without waking firstmate. -A stable true fires the action at most once. -One terminal outcome is then durably captured and published as a wake, which remains eligible for re-announcement until handled. - +Re-arm never acquires, releases, or hands off the source claim, and it may carry `--agent-reply-file ` whose contents are copied into that generation's own private staging file and handed once to the published `--agent-reply` argument; a re-arm that fails leaves the prior registration and the reply it references exactly as they were, including when the acknowledgement it owes cannot be recorded. +Posting that reply is best effort by design: the listener consumes the staged file only once its own setup and the board artifact have checked out, so the one loss window is a rare crash between that consume and the call it feeds, which drops that round's reply rather than posting it twice, and nothing here keeps a receipt, retry, or idempotency record - robust reply delivery waits on lavish-axi's exclusive listener. +The captured result is stored with immutable task-owner routing evidence and delivered directly to that task's steering inbox, without a firstmate `check` wake for the captain's words. +Filing that steering note away is not acknowledging the round, so while the round stays open every reconcile puts a live note back in the owner's inbox rather than ringing a filed one. +A task-owned source with an unhandled capture is not relaunched, so delivery failure cannot consume a round and start another poll. +That record is the only ownership evidence there is, so while any captured round of it is unacknowledged every retirement path refuses - the runner's own terminal retirement and an explicit `retire` alike - and the refusal names the acknowledgement that releases it. +A terminal result, including `session_ended`, an empty End, or missing, is delivered to the owner with an explicit stop-and-conclude instruction and is never auto-rearmed. +That round keeps the board with its owner: the source record is not retired while the terminal capture is unacknowledged, so no second armer can take the board, and acknowledging it with `bin/fm-procevent.sh handled ` is what concludes and retires it. +That conclude retains the registration it is retiring, removes it, then records the acknowledgement and restores the registration if that record cannot be written, so a failed conclude never leaves the round open with its owner gone. +An interruption between those two durable steps leaves the board unregistered with its terminal round still open, which nothing relaunches and the same `handled` call finishes. +It concludes only a round that is still open, so a repeated acknowledgement of an already-closed round reports `already-handled` and never touches whatever registration holds the board by then. +A second armer is refused with the current owner named, and the source list derives `listening`, `round-open`, or `dead` from the claim and handled captures without a second ownership record. +If the hosting worker cannot be recovered, relaunch a worker to re-host first; guarded firstmate adoption is an explicit last resort only after the old claim is proved dead. +The cross-home gap between worker rounds remains an accepted residual until lavish-axi's exclusive listener lands. +The interim crew instruction emitted by `bin/fm-brief.sh` points workers at this arm-and-acknowledge contract. + +The `when` adapter (`bin/fm-procevent-when.sh`) turns this channel into a condition->action primitive: it registers a deterministic condition and a deterministic action once, its blocking child polls the condition without waking firstmate, and a stable true fires the action at most once before one terminal outcome is durably captured and published as a wake that remains eligible for re-announcement until handled. The (condition, action) spec is stored privately under `state/when/` and hash-bound by a trust record the same way `bin/fm-check-register.sh` binds a custom check, while the spec separately binds the resolved action executable's bytes; a mutated or unregistered spec or a changed action executable is refused before the action runs, and that binding is reloaded from disk immediately before each fire rather than trusted from when polling started. A repo update that fast-forwards an in-repo action's bytes in place would otherwise desync every already-armed watch's trust binding with no tampering involved; `bin/fm-procevent-when.sh rebind-all` re-hashes and republishes the binding for every registered watch whose action lives under `FM_ROOT`, including one already polling, so it keeps firing across such an update instead of being refused on its next fire. - Every failure path - a mutated spec or action executable, a condition error past its budget, an expired deadline, a failed action, or an earlier fire whose outcome was never captured - produces a terminal captured outcome that wakes firstmate rather than a silent retry, and a durable single-fire marker claimed before the action makes restarts and re-polls unable to fire it twice. The adapter automates only the exact deterministic subset: anything needing judgment, and anything destructive, irreversible, or security-sensitive, keeps the ordinary check-fires-then-firstmate-decides flow, and the adapter's header and `--help` own its commands, flags, and outcome document. -**Capture and publish results** - This section is the single owner of the runner's operating contract. - -- Process-event commands resolve the state root to its physical directory before validating it and deriving paths, so a home reached through a symlinked ancestor behaves like its physical spelling while an unsafe target directory remains refused. -- Registration writes one private record under `state/procevent/`, and a completed result plus its immutable adapter identity are captured under `state/procevent-inbox/` before any announcement or event can reference it. -- By default, results are published as ordinary `check` wakes carrying the source id and committed result sequence through the existing durable wake queue, so the runner adds no second notification control plane. -- The self-announcing adapter exception and its fail-safe ordering are defined below. -- The watcher delivers a queued result on its ordinary cycle by reporting it as an actionable `check` wake, so a default or fallback publication reaches firstmate through the same rewake path every other wake uses and never waits for a manual drain. -- A queued `check` delivery is reported at most once per captured source and sequence while any records for that key remain queued. -- A durable handled acknowledgement stops future source re-announcement, while a record already queued remains under the durable queue's authority until the ordinary drain's sequence-bound post-handling acknowledgement consumes it. - -**Reconcile sources** +Process-event commands resolve the state root to its physical directory before validating it and deriving paths, so a home reached through a symlinked ancestor behaves like its physical spelling while an unsafe target directory remains refused. +Registration writes one private record under `state/procevent/`, and a completed result plus its immutable adapter identity are captured under `state/procevent-inbox/` before any announcement or event can reference it. +By default, results are published as ordinary `check` wakes carrying the source id and committed result sequence through the existing durable wake queue, so the runner adds no second notification control plane. +The self-announcing adapter exception and its fail-safe ordering are defined below. +The watcher delivers a queued result on its ordinary cycle by reporting it as an actionable `check` wake, so a default or fallback publication reaches firstmate through the same rewake path every other wake uses and never waits for a manual drain. +A queued `check` delivery is reported at most once per captured source and sequence while any records for that key remain queued. +A durable handled acknowledgement stops future source re-announcement, while a record already queued remains under the durable queue's authority until the ordinary drain's sequence-bound post-handling acknowledgement consumes it. Discovery is never a timer. -Each registered source has its own child process blocking on that source. -On every cycle, the watcher's `reconcile`: - -- Republishes every captured result without a durable handled acknowledgement, regardless of earlier publication. -- Restarts a source whose owner is gone. -- Stops this home's runner if its registration disappeared unexpectedly. - +Each registered source has its own child process blocking on that source, and the watcher's per-cycle `reconcile` republishes every captured result with no durable handled acknowledgement yet - regardless of any earlier publication - restarts a source whose owner is gone, and stops this home's runner when reconciliation runs after its registration disappeared unexpectedly. In supported steady state, a home with no registered source runs nothing, generates no state, and keeps its ordinary cadence. -**Suppress only adapter-confirmed no-op results** - Whether a captured result is a routine no-op is adapter knowledge too, and the runner names no adapter-specific condition for it either. - -- Before publishing, the runner asks the immutable captured owner through the built-in `silent` command or external `result.silent` operation and treats exit 0 as the only silence verdict: the result is recorded as durably handled and never announced, so it neither wakes a handler now nor returns on a later reconcile. -- The task-owned terminal exception is evaluated first, so an empty terminal board round goes to its owner's steering inbox for the required conclusion instead of entering this generic silence path. -- A missing command, an error, any other exit, or a silence the runner cannot durably record all publish the `check` wake exactly as before, so an adapter with no notion of a no-op needs no change and an unknown or degraded result always reaches its handler. -- For built-ins, silence remains independent of the keyed-answer feed below: suppressing an announcement never suppresses the captain's own answer. - -**Lavish silence rules** - -- For Lavish that verdict covers two shapes - a session the adapter classifies `ended` that carries no queued content block at all, which is a review surface closed with nothing said, and `browser_disconnected` (classified `disconnected`), which carries no answer while the session remains open. -- Any recognized top-level `prompts` or `feedback` block counts as content regardless of its declared count, and a malformed header makes the result indeterminate rather than empty. -- A `Send & End` close carrying the captain's answer arrives as `status: feedback` with `session_ended`, so it classifies `feedback` and is announced unchanged, as is any `ended` result that still carries content, and every `waiting`, `missing`, `unknown`, or unreadable result. - -**Retire terminal sources** +Before publishing, the runner asks the immutable captured owner through the built-in `silent` command or external `result.silent` operation and treats exit 0 as the only silence verdict: the result is recorded as durably handled and never announced, so it neither wakes a handler now nor returns on a later reconcile. +The task-owned terminal exception is evaluated first, so an empty terminal board round goes to its owner's steering inbox for the required conclusion instead of entering this generic silence path. +A missing command, an error, any other exit, or a silence the runner cannot durably record all publish the `check` wake exactly as before, so an adapter with no notion of a no-op needs no change and an unknown or degraded result always reaches its handler. +For built-ins, silence remains independent of the keyed-answer feed below: suppressing an announcement never suppresses the captain's own answer. +For Lavish that verdict covers two shapes - a session the adapter classifies `ended` that carries no queued content block at all, which is a review surface closed with nothing said, and `browser_disconnected` (classified `disconnected`), which carries no answer while the session remains open. +Any recognized top-level `prompts` or `feedback` block counts as content regardless of its declared count, and a malformed header makes the result indeterminate rather than empty. +A `Send & End` close carrying the captain's answer arrives as `status: feedback` with `session_ended`, so it classifies `feedback` and is announced unchanged, as is any `ended` result that still carries content, and every `waiting`, `missing`, `unknown`, or unreadable result. Whether a captured result ends its source is adapter knowledge, never the runner's. -After capture, the runner asks the immutable captured owner whether the result is terminal. -It uses the built-in `terminal` command or external `result.terminal` operation. -Under the default ordering, this happens after the initial `check` publication. - -- Exit 0 retires the registration. - The exception is a task-owned board, whose owner must first acknowledge the round as defined above. -- Retirement drops only the exact registration generation captured by the claim. - Under one source boundary, it releases that claim only after removal succeeds. -- A missing command, an error, or any other exit keeps the source armed. - An adapter with no notion of ending needs no change. - -- A failed terminal removal stays durably terminal and is completed by ordinary reconciliation without restarting its poll, while a concurrently replaced registration survives and becomes independently runnable after the old claim releases. -- Any registration refuses to replace an external registration while its prior runner claim is live, uncertain, orphaned, or terminal-pending; replacement becomes eligible only after that generation is proved gone or its terminal retirement completes. -- A source that has ended therefore captures at most one terminal result, is never restarted, and leaves no recurring poll work. -- For ordinary sources, explicit `retire` stays the supported and idempotent path afterwards; a task-owned board instead refuses `retire` until its owner concludes the open terminal round with `handled`. -- For Lavish that verdict covers an ended session, a missing session, and the final feedback of a `Send & End` review, which the published poll marks with `session_ended` before it returns only empty ended sessions. - -**Apply built-in results automatically** +After capture - and after initial `check` publication for the default ordering - the runner asks the immutable captured owner through the built-in `terminal` command or external `result.terminal` operation and retires the registration on exit 0 alone - except a task-owned board, whose terminal retirement is refused until its owner acknowledges the round, as the crew-hosted section above defines - dropping only the exact registration generation captured by its claim and releasing that claim only after removal succeeds under one source boundary; a missing command, an error, or any other exit keeps the source armed, so an adapter with no notion of ending needs no change. +A failed terminal removal stays durably terminal and is completed by ordinary reconciliation without restarting its poll, while a concurrently replaced registration survives and becomes independently runnable after the old claim releases. +Any registration refuses to replace an external registration while its prior runner claim is live, uncertain, orphaned, or terminal-pending; replacement becomes eligible only after that generation is proved gone or its terminal retirement completes. +A source that has ended therefore captures at most one terminal result, is never restarted, and leaves no recurring poll work. +For ordinary sources, explicit `retire` stays the supported and idempotent path afterwards; a task-owned board instead refuses `retire` until its owner concludes the open terminal round with `handled`. +For Lavish that verdict covers an ended session, a missing session, and the final feedback of a `Send & End` review, which the published poll marks with `session_ended` before it returns only empty ended sessions. Applying a captured result through code is a built-in adapter seam, and some built-in results carry no judgement at all: they must simply be applied idempotently to this home's own durable state. Leaving that to a handler means it can silently not happen, so immediately after the terminal check above the runner calls `bin/fm-procevent-.sh autohandle ` and lets the built-in adapter apply and acknowledge its own result. - That call runs strictly after terminal retirement, because a handling adapter re-arms its own next source and retiring afterwards would drop that fresh registration and leave the source silently dead. Exit 0 means the adapter fully applied and acknowledged the result; a missing command, an error, or any other exit is not a capture failure but leaves the result unacknowledged and therefore still eligible for re-announcement, so a handler receives it exactly as before and an adapter with no such command needs no change. - -**Adapter-controlled announcement order** - -The built-in `bin/fm-procevent-.sh self-announcing` command declares announcement order: - -| Response | Runner behavior | -| --- | --- | -| Exit 0 | The adapter declares that every result its autohandle fully applies is announced through its own durable downstream channel; the runner applies first, then publishes a `check` wake only for results still unhandled. | -| Any other response | Keep strict publish-before-apply ordering; autohandle runs only after this capture's own wake was successfully appended to the durable queue. | - +Announcement ordering is adapter-declared through `bin/fm-procevent-.sh self-announcing`: an adapter that answers exit 0 declares that every result its autohandle fully applies is announced through a durable downstream channel of its own, so the runner applies first and publishes a `check` wake only for what remains unhandled afterwards; every other adapter keeps the strict publish-before-apply order, and its autohandle runs only when this capture's own wake was successfully appended to the durable queue. The remote-secondmate reply adapter declares itself self-announcing: a captured reply reaches its local status mirror and settles its correlated pending-reply expectation without any handler step, the mirrored status bytes are the single wake for one remote note through the same signal classification a local secondmate's append gets, and only a capture the adapter could not fully apply is published as a `check` wake, whose adapter handling remains idempotent. The [remote-secondmate channel contract](remote-secondmates.md#normal-operation) owns replay suppression and its bounded upgrade exception; a replay that adds no mirror bytes stays quiet. -**Feed keyed captain answers** - Keyed captain answers from built-in adapters use one more seam of the same kind, and the runner still decides nothing about them. Some built-in sources carry the captain's answer to a captain-held task, and what such an answer means is owned once by `bin/fm-captain-hold.sh`'s keyed-answer intake rather than by any channel. - -- A built-in source bound with `bin/fm-captain-hold.sh bind` therefore has each captured result passed to `bin/fm-procevent-.sh answers `, and whatever that prints is piped straight into that intake. -- A binding can select one decision origin or the script's cross-origin mode; the command header owns the exact forms and key interpretation. -- The built-in adapter reports only what the captain chose; the intake owns every rule about what happens next, so the runner names no adapter, parses no result, and carries no decision rule, and a future built-in answer source needs nothing here beyond an `answers` command and a binding. - -**Reconcile selections and handling boundaries** - -- The reserved Reconcile selection uses the parallel optional `reconciles` adapter command and binding-verified `reconcile-requests` intake rather than entering keyed answers; [`captain-hold-lifecycle.md`](captain-hold-lifecycle.md#reconcile-re-check-reality-never-a-blind-close) owns those semantics. -- Feeding is independent of handling: it never acknowledges a result and never suppresses a wake, because recording the answer or request is transcription while acting on it is firstmate's judgement. -- An unbound built-in source, a built-in adapter without the corresponding command, and a failure on either side all leave the capture untouched and still announced. -- External binding responses never enter either authority-bearing intake. - -**Machine-wide source ownership** +A built-in source bound with `bin/fm-captain-hold.sh bind` therefore has each captured result passed to `bin/fm-procevent-.sh answers `, and whatever that prints is piped straight into that intake. +A binding can select one decision origin or the script's cross-origin mode; the command header owns the exact forms and key interpretation. +The built-in adapter reports only what the captain chose; the intake owns every rule about what happens next, so the runner names no adapter, parses no result, and carries no decision rule, and a future built-in answer source needs nothing here beyond an `answers` command and a binding. +The reserved Reconcile selection uses the parallel optional `reconciles` adapter command and binding-verified `reconcile-requests` intake rather than entering keyed answers; [`captain-hold-lifecycle.md`](captain-hold-lifecycle.md#reconcile-re-check-reality-never-a-blind-close) owns those semantics. +Feeding is independent of handling: it never acknowledges a result and never suppresses a wake, because recording the answer or request is transcription while acting on it is firstmate's judgement. +An unbound built-in source, a built-in adapter without the corresponding command, and a failure on either side all leave the capture untouched and still announced. +External binding responses never enter either authority-bearing intake. Ownership is machine-wide per canonical source, because separate homes can share one underlying source store. - -- Claims live under `$XDG_STATE_HOME/firstmate/procevent-claims` (override with `FM_PROCEVENT_CLAIM_ROOT`). -- Each claim binds its caller-reported home and runner PID to a process identity, unique claim generation, exact registration-file generation, and resolved state-root identity. -- Registration, acquisition, replacement, retirement, and generation-bound release are serialized at one machine-wide boundary per source. -- A live identity-matched owner is never displaced, and release removes only the exact generation the caller acquired. - -**Prove ownership before stopping a runner** - -- Every stop proves ownership before its first signal: the live runner's recorded process identity must match and it must still lead its process group. -- Once that stop has proved ownership and sent TERM, its own escalation to KILL checks only whether the proved group still has members; it does not re-read the leader's identity or group membership, which can change or become unreadable as TERM ends the leader. -- This proof belongs only to that stop's own escalation and cannot authorize another caller that encounters an unproved group. - -**Recover orphaned claims** - +Claims live under `$XDG_STATE_HOME/firstmate/procevent-claims` (override with `FM_PROCEVENT_CLAIM_ROOT`). +Each claim binds its caller-reported home and runner PID to a process identity, unique claim generation, exact registration-file generation, and resolved state-root identity. +Registration, acquisition, replacement, retirement, and generation-bound release are serialized at one machine-wide boundary per source. +A live identity-matched owner is never displaced, and release removes only the exact generation the caller acquired. +Every stop proves ownership before its first signal: the live runner's recorded process identity must match and it must still lead its process group. +Once that stop has proved ownership and sent TERM, its own escalation to KILL checks only whether the proved group still has members; it does not re-read the leader's identity or group membership, which can change or become unreadable as TERM ends the leader. +This proof belongs only to that stop's own escalation and cannot authorize another caller that encounters an unproved group. A stale claim whose process group still has members is one `reconcile` never displaces, and the two shapes it comes in recover differently. `reconcile` preserves such a claim without signalling the ambiguous group or starting a replacement: the group check probes the runner's own process group, which contains its polling source child, so surviving members can mean that child is still attached to the session the source collects from, and a replacement would put a second destructive poller on it. - `list` reports both shapes as `orphaned`. - -**Reused PID with surviving group members** - -When the recorded pid is alive under a different identity but the group still has members, the claim boundary itself does not consult the process group. -In this case, `bin/fm-procevent.sh start ` reclaims the claim only if it can tidy the dead generation's reservation records. -Otherwise it refuses with `cannot claim source`. - -Tidy-up is waived only for a generation proved gone. -This generation does not meet that condition. +When the recorded pid is alive under a different identity while the group still has members, the claim boundary itself does not consult the process group, so `bin/fm-procevent.sh start ` reclaims that claim provided the dead generation's reservation records can still be tidied, and otherwise refuses with `cannot claim source`; that tidy-up is waived only for a generation proven gone, which this one is not. That hand-run command is the recovery path, taken by someone who has checked that nothing is still polling the source. - That asymmetry between the automatic path and the deliberate one is the design rather than an inconsistency, and it is not a claim-level invariant: nothing below `reconcile` enforces it. - -**Dead leader with surviving group members** - -If the leader died from anything other than the stop's own signal and its group still has members, `start` does not reclaim the claim. -It reports `already owned` and changes nothing. - -`retire`, `reconcile`, `sweep-home`, and the guard all refuse the surviving group permanently, so the source stops listening. +When the leader itself is gone and its group still has members - the leader died to anything other than the stop's own signal - `start` does not reclaim the claim either: it reports `already owned` and changes nothing, and `retire`, `reconcile`, `sweep-home`, and the guard all refuse the surviving group permanently, so the source stops listening. Recovery there is a human verifying whether the dead runner's polling child is still attached to the source; once that process group is empty the generation reads as gone and the next `reconcile` reclaims the source on its own. - Nothing automatic signals that group, and whether it may ever be signalled remains an open decision; the repaired guard does not close this gap. - -**Report stranded claims** - -The first `reconcile` that strands either kind of claim generation publishes a durable `check` wake. -It names the source and the recovery step: - -- For a reused pid, the `start` command. -- For a leaderless group, the check to make. - -Later cycles stay silent for the same generation. -A genuinely new stranded claim announces again. - -**Reclaim a generation proved gone** - +Neither shape stops listening quietly: the first `reconcile` that strands a claim generation publishes a durable `check` wake naming the source and what clears it - the `start` command for the reused pid, the check to make for the leaderless group - and later cycles stay silent for that same generation while a genuinely new stranded claim announces again. Reclaiming a generation that IS gone is not gated on tidying anything that generation left behind: its capture-reservation records, its staging file, or the registry directory a claim recorded for them. Every one of those is keyed by claim token and every replacement claims a fresh one, so a leftover that can no longer be located or removed - a state-root identity a claim recorded before its home was re-created, or a recorded registry directory that no longer resolves to a directory - is stale bytes rather than an ownership hazard. - Making any of them a precondition is what leaves a provably dead runner owning its source permanently, because none of those conditions clears on its own. - -- Ordinary release and reclamation still attempt reservation cleanup and require it unless both owner staleness and whole-group absence prove the generation gone. -- The narrow live-owner terminal-self-retirement path also attempts cleanup but tolerates its own still-in-flight reservation, which the runner removes on the normal end-of-capture path; exact home, PID, and claim-token ownership remains mandatory before the claim is released. - -**Stop refusals and residual races** - -- If identity cannot be established before the first signal, or a surviving owned group cannot be proved stopped, the operation preserves the registration and claim for safe retry rather than adding a second owner. -- A live PID whose identity no longer matches is refused before the first signal. -- Identity and process-group verification cannot be made atomic with signalling in portable shell: the reaper signals only a target it has verified as the recorded generation, but PID and group reuse remain possible in the narrow interval between verification and the signal. -- Launch pacing is the primary host-wedge protection; watchdog cleanup is a backstop. - -**Retire a secondmate home** - -- Supported secondmate retirement preflights each target home's bounded `sweep-home` command before destructive teardown, snapshots its registrations outside the target, then runs the sweep at that home's final deletion or return boundary. -- If deletion or return fails, teardown restores those registrations and reconciles them before returning the refusal. -- If restoration or rearming also fails, teardown returns a distinct status and reports the retained registration backup path for manual recovery instead of hiding the retired waits. -- The sweep retires local registrations and machine-wide claims whose recorded state-root identity matches that home's resolved state root through the same identity-checked, generation-bound retirement path, and leaves foreign-home claims untouched. -- Teardown refuses with the home, lease, routing evidence, registrations, claims, and runners retained when identity is uncertain, ownership is unreadable or unreleased, or relevant state exists without a sweep-capable child script. - -**Recover from unsupported manual deletion** - +Ordinary release and reclamation still attempt reservation cleanup and require it unless both owner staleness and whole-group absence prove the generation gone. +The narrow live-owner terminal-self-retirement path also attempts cleanup but tolerates its own still-in-flight reservation, which the runner removes on the normal end-of-capture path; exact home, PID, and claim-token ownership remains mandatory before the claim is released. +If identity cannot be established before the first signal, or a surviving owned group cannot be proved stopped, the operation preserves the registration and claim for safe retry rather than adding a second owner. +A live PID whose identity no longer matches is refused before the first signal. +Identity and process-group verification cannot be made atomic with signalling in portable shell: the reaper signals only a target it has verified as the recorded generation, but PID and group reuse remain possible in the narrow interval between verification and the signal. +Launch pacing is the primary host-wedge protection; watchdog cleanup is a backstop. + +Supported secondmate retirement preflights each target home's bounded `sweep-home` command before destructive teardown, snapshots its registrations outside the target, then runs the sweep at that home's final deletion or return boundary. +If deletion or return fails, teardown restores those registrations and reconciles them before returning the refusal. +If restoration or rearming also fails, teardown returns a distinct status and reports the retained registration backup path for manual recovery instead of hiding the retired waits. +The sweep retires local registrations and machine-wide claims whose recorded state-root identity matches that home's resolved state root through the same identity-checked, generation-bound retirement path, and leaves foreign-home claims untouched. +Teardown refuses with the home, lease, routing evidence, registrations, claims, and runners retained when identity is uncertain, ownership is unreadable or unreleased, or relevant state exists without a sweep-capable child script. Raw manual deletion of a Firstmate home is unsupported because it can orphan a blocking child. To recover, restore that home's tracked `bin/fm-procevent.sh`, run `FM_HOME= /bin/fm-procevent.sh sweep-home`, then rerun the supported teardown. - The owning-home lease below bounds how long such an orphan can run, but it is a backstop, not a substitute for the supported path. -**Home lease and its limits** - A runner is bound to the HOME that owns it, not to the one session that armed it. That granularity is deliberate: a persistent source is meant to outlive the turn and the session that armed it, so binding a runner to its arming session would stop exactly the sources this mechanism exists to keep running. - Any activity in the same home refreshes the lease, so a replacement session, another watcher, or an ordinary inspection command keeps a runner of that home alive; a runner whose SOURCE is no longer wanted in a live home is stopped by reconcile when that source is retired, independently of the lease. The lease is therefore the backstop for a home that is GONE - the torn-down test sandbox this change exists to bound - and not a per-session ownership check. - KNOWN LIMIT: while any activity continues in a home whose original owning session has ended, that activity refreshes the lease and a runner of that home keeps running until its source is retired or the home goes away. - -**Keep and guard the lease** - Detaching a runner into its own process group is what lets a persistent source outlive the turn that armed it, and on its own it is also what lets a runner outlive its whole home: reparented to init, it keeps its blocking child - and every process that child spawns - running with nothing left to reap it. - -- So a home's process-event state carries a lease that registration, attached start, reconciliation, acknowledgement, and listing refresh, and the watcher's reconcile cycle is what keeps it fresh in a live home. -- An attached public `start` continues refreshing the lease while its caller remains attached. -- Each runner fails closed unless a small guard starts successfully beside it in a separate process group. -- That guard accepts the lease only while the state root retains the device/inode identity recorded by the runner's claim, and initiates the verified stop after two consecutive reads cannot prove that identity and lease freshness, so one unreadable read cannot kill a live runner. -- Those two reads are spaced half a check interval apart, so the pair the debounce requires completes inside one check interval instead of costing two of them. - -**Detection and stop timing** - -For a runner whose ownership can still be proved, the nominal detection bound is the lease plus one check interval. -The verified stop then runs within its own grace period. -The lease age is compared in whole seconds, so the configured lease is honoured until that age reads one second past it. - -Scheduling delays or failed inspection and signalling can extend the whole bound. +So a home's process-event state carries a lease that registration, attached start, reconciliation, acknowledgement, and listing refresh, and the watcher's reconcile cycle is what keeps it fresh in a live home. +An attached public `start` continues refreshing the lease while its caller remains attached. +Each runner fails closed unless a small guard starts successfully beside it in a separate process group. +That guard accepts the lease only while the state root retains the device/inode identity recorded by the runner's claim, and initiates the verified stop after two consecutive reads cannot prove that identity and lease freshness, so one unreadable read cannot kill a live runner. +Those two reads are spaced half a check interval apart, so the pair the debounce requires completes inside one check interval instead of costing two of them. +For a runner whose ownership can still be proved, the nominal detection bound is therefore the lease plus one check interval, after which the verified stop runs within its own grace period; the lease age is compared in whole seconds, so a configured lease is honoured until that age reads one second past it, and scheduling delays or failed inspection and signalling can extend the whole bound. That grace is a ceiling rather than a delay every stop pays: two seconds for the ordinary signal and two more for the forced one, spent only by a group that outlives the signal it was sent, which is why a healthy runner's stop completes in a fraction of a second. - The group signal reaches the blocking child and everything under it exactly as retirement does. - -**Prevent accidental self-refresh** - -- A runner exports the inherited `FM_PROCEVENT_IN_RUNNER` marker and every lease refresh is skipped under it, so a runner and its ordinary children do not certify their own owner, and the next reconcile in a live home simply starts a replacement runner. -- That no-self-refresh rule is CONFUSED-AGENT-GRADE, the same deliberate captain-decided grade `bin/fm-lease-lib.sh` documents: it stops the accidental case this boundary exists for, an orphaned or test-scaffolding source tree that would otherwise keep its own owner alive. -- A source that DELIBERATELY strips the marker from its environment can still refresh the lease, so adversarial-grade unforgeability is explicitly out of scope here and tracked as separate follow-up design work. -- Scope is the owning state root and one runner generation, never a script or process name, so a live source in another home is untouched: that home refreshes its own lease. - -**Lease and launch pacing settings** - -| Setting | Default | Range | Purpose | -| --- | --- | --- | --- | -| `FM_PROCEVENT_OWNER_LEASE_SECONDS` | 600 | 1..86400 | How long a runner continues without activity in its owning home. | -| `FM_PROCEVENT_OWNER_CHECK_SECONDS` | 15 | 1..3600 | Guard detection interval; it reads the lease and recorded state-root identity twice per interval, half an interval apart, so both debounce reads fit inside one interval. | -| `FM_PROCEVENT_LAUNCH_FLOOR_SECONDS` | 1 | 1..3600 | Minimum time between consecutive launches of one registration generation's stored command; bounds immediately returning sources during the lease window. | - +A runner exports the inherited `FM_PROCEVENT_IN_RUNNER` marker and every lease refresh is skipped under it, so a runner and its ordinary children do not certify their own owner, and the next reconcile in a live home simply starts a replacement runner. +That no-self-refresh rule is CONFUSED-AGENT-GRADE, the same deliberate captain-decided grade `bin/fm-lease-lib.sh` documents: it stops the accidental case this boundary exists for, an orphaned or test-scaffolding source tree that would otherwise keep its own owner alive. +A source that DELIBERATELY strips the marker from its environment can still refresh the lease, so adversarial-grade unforgeability is explicitly out of scope here and tracked as separate follow-up design work. +Scope is the owning state root and one runner generation, never a script or process name, so a live source in another home is untouched: that home refreshes its own lease. +`FM_PROCEVENT_OWNER_LEASE_SECONDS` (default 600, range 1..86400) is how long a runner keeps going with no sign of activity in its owning home, and `FM_PROCEVENT_OWNER_CHECK_SECONDS` (default 15, range 1..3600) is the guard's detection interval: it re-reads the lease and the recorded state-root identity twice within each interval, half an interval apart, so the two reads its debounce needs fit inside one interval rather than costing two. +`FM_PROCEVENT_LAUNCH_FLOOR_SECONDS` (default 1, range 1..3600) is the minimum time between consecutive launches of one registration generation's stored command, bounding the launch rate of an immediately returning source during that lease window. The generation's first launch is immediate, later launches share its monotonic pacing timestamp, a timestamp from before a reboot is treated as expired, and replacing the registration starts a fresh pacing generation. -**Confirm detached launches** - `FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS` (default 3, range 1..600) bounds how long `reconcile` waits for the runners it just started to prove they are running: never less than the configured value, and at most one second more, because the wait is measured on a whole-second clock. - -- Starting a runner is detached and its errors are not visible to the caller, so `reconcile` reports a start only after the source is observed owned or its launch-pacing stamp has advanced or appeared, and reports every unconfirmed launch as `failed=` and a non-zero exit instead. -- Both signals are durable evidence a runner claimed: ownership is the only evidence a runner still blocked on its source ever shows, and the stamp - written after the claim and before the source command runs, and removed only by registration replacement - covers a runner that claimed, ran and exited between two polls. -- A healthy launch therefore confirms on the first poll and the window only bounds a launch that has not yet proved itself - one that died before claiming, or one merely too slow to claim inside the window; confirmation cannot tell those apart, and a launch that proves itself on a later cycle closes its failure episode without a retraction wake. -- All of a cycle's launches share one window, so a home full of sources that cannot start costs the same bounded wait as one. - -**Keep confirmation below the watcher interval** +Starting a runner is detached and its errors are not visible to the caller, so `reconcile` reports a start only after the source is observed owned or its launch-pacing stamp has advanced or appeared, and reports every unconfirmed launch as `failed=` and a non-zero exit instead. +Both signals are durable evidence a runner claimed: ownership is the only evidence a runner still blocked on its source ever shows, and the stamp - written after the claim and before the source command runs, and removed only by registration replacement - covers a runner that claimed, ran and exited between two polls. +A healthy launch therefore confirms on the first poll and the window only bounds a launch that has not yet proved itself - one that died before claiming, or one merely too slow to claim inside the window; confirmation cannot tell those apart, and a launch that proves itself on a later cycle closes its failure episode without a retraction wake. +All of a cycle's launches share one window, so a home full of sources that cannot start costs the same bounded wait as one. Keep this window well below `FM_POLL`. `bin/fm-watch.sh` runs `reconcile` once per supervision cycle, so a source that cannot start makes every cycle wait up to the confirm window before the rest of that cycle runs. - Raising the confirm window lengthens every supervision cycle and delays wake delivery by up to that much. -**Report launch failures** - A source that can never start is reported as `failed=` with a non-zero exit on every `reconcile`, rather than counted as `started` and retried silently as though it were healthy, so a wedged source stays visible instead of presenting as armed. -The `failed=` count reaches only the command's caller because `bin/fm-watch.sh` discards `reconcile` output and exit status. -For that reason, `reconcile` also publishes a durable `check` wake once per failure episode, with key `procevent::launch-failed:-`. -Later cycles stay silent for that episode until a launch confirms. -A later fresh failure gets a fresh key, because the watcher never re-surfaces a key it has already surfaced. - -- The announcement changes nothing about the launch: `reconcile` keeps relaunching the source every cycle exactly as before, and nothing is retried differently, throttled, or recovered from that signal. -- The wake reports only the observed failure: the launch did not prove that it took the claim within the window. -- If the failure persists, inspect the source command and adapter binary named in the registration. - The wake names both, along with the attached `bin/fm-procevent.sh start ` command that reproduces the refusal on stderr. - The detached launch discards that output. -- A later cycle that finds the source owned ends the episode automatically. - A runner that was merely slow to claim needs no operator action. -- A source stranded on a claim nothing may automatically displace is announced the same way, once per stranded claim generation, as described above. -- `bin/fm-watch.sh` surfaces both under their own headlines - `process-event source stranded` and `process-event source failed to start` - rather than as a captured result. - -**Reject unusable settings** +That count reaches only whoever runs the command, because `bin/fm-watch.sh` discards `reconcile`'s output and exit status, so an unconfirmed launch is also announced through the wake queue: `reconcile` publishes a durable `check` wake (`procevent::launch-failed:-`) once per failure episode, and later cycles stay silent for that episode until a launch of that source confirms, after which a fresh failure announces again under a fresh key, because the watcher never re-surfaces a key it has already surfaced. +The announcement changes nothing about the launch: `reconcile` keeps relaunching the source every cycle exactly as before, and nothing is retried differently, throttled, or recovered from that signal. +The wake says only what was observed for that shape - the launch did not prove it took the claim within the window - and, if it stays that way, names the source command and adapter binary the registration names as what to check and the attached `bin/fm-procevent.sh start ` as what reproduces a refusal on stderr, where the detached launch discards it; a later cycle that finds the source owned ends the episode on its own, so a runner that was merely slow to claim needs nothing from the operator. +A source stranded on a claim nothing may automatically displace is announced the same way, once per stranded claim generation, as described above. +`bin/fm-watch.sh` surfaces both under their own headlines - `process-event source stranded` and `process-event source failed to start` - rather than as a captured result. A value this command cannot use is refused by name before anything is launched, the same way `FM_PROCEVENT_LAUNCH_FLOOR_SECONDS` and `FM_PROCEVENT_MAX_OUTPUT_BYTES` are refused, so a mistyped window can never present as a fleet of sources that cannot start. `bin/fm-watch.sh` validates the same value when it arms and refuses to arm on an unusable one, naming the variable and the range: under a running watcher that refusal would otherwise repeat on every cycle into a discarded stdout and leave the whole home disarmed while presenting as supervised, whereas a watcher that will not arm is loud through the liveness guard. -**Limit captured output** - `FM_PROCEVENT_MAX_OUTPUT_BYTES` (default 1048576) bounds a single captured result while the source runs; oversized output is drained but truncated with a stderr notice rather than staged or published whole or dropped. -**Durability guarantees and limits** - The runner proves exactly one durability boundary: output that reached the runner is stored at mode `0600` before any event referencing it is published, and a captured result with no durable handled acknowledgement remains eligible for bounded re-announcement across any number of drains and restarts, not only the crash window right after capture. - -- `bin/fm-procevent.sh handled ` is the only thing that stops re-announcement: a generation-keyed, private, path-safe, durable, and idempotent acknowledgement that atomically checks and deduplicates by the exact source and sequence, so a paired effect gated on its first-time-vs-repeat report is never authorized twice. -- Default and fallback `check` publication is still best-effort, so the same source and sequence can repeat even before any restart; handlers deduplicate that identity rather than assuming a wake is unique. -- The runner proves nothing about the source side, and the handled acknowledgement proves nothing about a paired external effect performed before it: a crash between that effect and the acknowledgement call can still repeat the effect on replay, so this is never a generic exactly-once guarantee. -- The published `lavish-axi poll` clears feedback destructively before returning it, so a result lost between that clearing and the runner reading process output is unrecoverable. -- Never describe this path as at-least-once, no-loss, or lossless. - +`bin/fm-procevent.sh handled ` is the only thing that stops re-announcement: a generation-keyed, private, path-safe, durable, and idempotent acknowledgement that atomically checks and deduplicates by the exact source and sequence, so a paired effect gated on its first-time-vs-repeat report is never authorized twice. +Default and fallback `check` publication is still best-effort, so the same source and sequence can repeat even before any restart; handlers deduplicate that identity rather than assuming a wake is unique. +The runner proves nothing about the source side, and the handled acknowledgement proves nothing about a paired external effect performed before it: a crash between that effect and the acknowledgement call can still repeat the effect on replay, so this is never a generic exactly-once guarantee. +The published `lavish-axi poll` clears feedback destructively before returning it, so a result lost between that clearing and the runner reading process output is unrecoverable. +Never describe this path as at-least-once, no-loss, or lossless. `docs/verification/process-event-sources.md` holds the measurements and `.agents/skills/process-event-sources/SKILL.md` owns the handling procedure. ## Spoken interface and captain inbox (config/voice-*, config/inbox-*) The spoken interface in [`docs/voice-relay.md`](voice-relay.md) and the model-backed subcommands of `bin/fm-inbox.sh` reach a paid API in a named account, so no region, model id or AWS profile is shipped as a tracked default. Each is one line in a local, gitignored `config/` file, with an environment variable that overrides it for a single run, and a missing required value refuses with the path to write rather than falling back to a value that belongs to another home. - That configuration is the whole opt-in: an unconfigured home cannot start the relay and cannot run `fm-inbox.sh say` or `ask`, while `note`, `announce`, `reply`, `receipts`, `ready`, `status`, `list` and `drain` need no configuration at all because they make no model call. The voice handover depends on `note`, so it keeps working in a home that has configured nothing. @@ -2185,15 +1092,8 @@ The voice handover depends on `note`, so it keeps working in a home that has con | `config/inbox-ask-model` | `FM_INBOX_ASK_MODEL` | Side-question model id, required by `fm-inbox.sh ask`. | | `config/inbox-profile` | `FM_INBOX_PROFILE` | AWS profile for those two calls; absent, or an explicitly empty variable, means whatever credentials are already in the environment. | -**How configuration files are parsed** - Each account, model and voice file above is read as its first line that is not blank and not a `#` comment, so a comment above the value is fine. -The two read files use different parsing rules: - -- `config/voice-read-scope` must contain only the bare word with optional blank space around it. - A comment header causes a refusal rather than being skipped. -- In `config/voice-read-deny`, every line that is neither blank nor a `#` comment adds one substring. - +The two read files are parsed differently: `config/voice-read-scope` must hold the bare word and nothing but blank space around it, so a comment header there refuses instead of being skipped, while every line of `config/voice-read-deny` that is not blank and not a `#` comment is one more substring. `FM_VOICE_RELAY` and `FM_VOICE_PYTHON` belong to the laptop rather than to a home, so they have no config file: `bin/fm-voice-client.py` requires the relay path as a flag or that variable and carries no default path. ## Environment variables @@ -2221,7 +1121,7 @@ FM_SESSION_START_QUEUED_LIMIT=20 # plain queued backlog rows in the session-st FM_BACKLOG_ROW_TIMEOUT_SECS=10 # seconds bounding each backlog row read (bin/fm-backlog-transition-lib.sh); nonpositive or invalid values fall back to 10; the first bound hit latches the sweep so later reads return immediately, each still naming its own item FM_BOOTSTRAP_DETECT_ONLY=0 # internal/read-only session-start mode: skip bootstrap's mutating sweeps and print advisory TANGLE wording FM_BOOTSTRAP_NETWORK=all # internal session-start phase split: all, skip (local steps only), or only (network steps only); see bin/fm-bootstrap.sh -FM_STARTUP_NETWORK_TIMEOUT=120 # seconds bounding the deferred inactive-outcome scan plus network checks, including the lock waits the worker makes before them; hitting it prints an actionable NETWORK_CHECKS line, and a lock a live process still holds at the deadline ends the worker with a failed-rerun record (publication and delivery are bounded by FM_SESSION_START_TIMEOUT the same way) +FM_STARTUP_NETWORK_TIMEOUT=120 # seconds bounding the deferred inactive-outcome scan plus network checks; hitting it prints an actionable NETWORK_CHECKS line FM_TASKS_AXI_COMPATIBLE= # internal one-hop handoff of an already-computed tasks-axi compatibility verdict (0 or 1); consumed when bin/fm-tasks-axi-lib.sh is sourced FM_GUARD_READ_ONLY=0 # internal/read-only guard mode: keep alarms but suppress drain, supervision repair, and checkout repair commands FM_GUARD_CONTINUE_LINE='This is a supervision warning only; the guarded operation WILL still run.' # banner continuation line; fm-send.sh overrides it to name the requested message specifically @@ -2259,7 +1159,6 @@ FM_PROCEVENT_LAUNCH_FLOOR_SECONDS=1 # minimum interval between launches of o FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS=3 # how long reconcile waits for the runners it started to prove they are running; 1..600, keep well below FM_POLL FM_WHEN_OUTPUT_TAIL_BYTES=8192 # bound on the command-output tail inside one condition->action outcome document FM_CODEX_WATCH_CHECKPOINT=180 # seconds per foreground watcher checkpoint in Codex primary supervision -FM_CODEX_WATCH_CHECKPOINT_AWAY=3600 # requested away checkpoint bound on a home with config/supervision-host; longer of this and attended bound, capped at 27000 FM_CREW_STATE_NM_TIMEOUT=10 # seconds allowed per no-mistakes query inside fm-crew-state.sh, and per state-database run-inventory read behind a capped AXI overview FM_TEARDOWN_NM_TIMEOUT=10 # seconds allowed per no-mistakes query or abort inside fm-teardown.sh FM_CREW_STATE_RUNS_LIMIT=200 # plain runs-ledger rows scanned for fallback attribution; does not change the CLI's AXI overview window (selection owner: bin/fm-nm-run-lib.sh) @@ -2299,7 +1198,6 @@ FM_WATCH_REARM_RETRY_LIMIT=5 # Pi/OpenCode adapter launch-failure retries befo FM_WATCH_CYCLE_LOG_MAX_BYTES=262144 # size cap for the arm-owned watcher lifecycle ledger FM_WATCH_CYCLE_LOG_KEEP_LINES=1000 # newest complete lifecycle rows considered when the ledger is capped FM_WATCHER_STALE_GRACE=300 # defaults to FM_GUARD_GRACE if set, else the poll-derived grace (docs/turnend-guard.md "Guard grace and the poll cadence"); seconds a live watcher lock may have a stale beacon before re-arm errors -FM_WATCHER_STALL_BOUND= # defaults to 3x FM_WATCHER_STALE_GRACE; a live holder whose beacon is stale past this hard bound is evicted with TERM and replaced by the re-arm rather than refused (docs/turnend-guard.md, bin/fm-watch.sh header) FM_SIGNAL_GRACE=30 # seconds to coalesce nearby status and turn-end signals into one wake FM_TURNEND_CHURN_ABSORB_SECS=900 # longest one endpoint's bare turn-ends may be deferred on pane-churn evidence alone; only consulted when config/turnend-churn-absorb is present FM_CAPTAIN_RE='done:|needs-decision:|blocked:|failed:|PR ready|checks green|ready in branch|merged' # captain-relevant status regex; nonterminal progress verbs remain excluded even when their prose matches @@ -2308,11 +1206,10 @@ FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stal FM_BUSY_TURN_MAX_SECS=3600 # maximum age without a completed turn or explicit native-harness progress (bin/fm-watch.sh owns marker selection), before the same wedge escalation used for a provably-working non-busy stale takes over; inspection-only, never an automatic interrupt or restart; a declared external wait, an attended verified captain-held transfer, or - where config/wedge-defer-parked-gate arms it - a validation gate of the crew's own awaiting the supervisor's still-unanswered decision takes the FM_PAUSE_RESURFACE_SECS recheck below instead FM_PAUSE_RESURFACE_SECS=14400 # four hours between bounded rechecks of a declared external wait or verified captain-held transfer, and between repeated new-hash stale alarms for an ordinary crew task with an open backlog captain call; a structured until time can make an external-wait recheck occur sooner but cannot extend this bound; this includes a live idle pane after its first inconclusive stale wake, a provably-working pane whose own unelapsed declared wait or, where config/wedge-defer-parked-gate arms it, unanswered supervisor-owed validation gate defers its FM_STALE_ESCALATE_SECS escalation, and a live busy pane past FM_BUSY_TURN_MAX_SECS, while the away-mode daemon uses the same setting and ages its window against the crew's own latest status line rather than pane busy state; a captain-held transfer is never rechecked while the away-posture record exists, while an armed validation gate awaiting the supervisor's decision keeps this recheck in either posture FM_SECONDMATE_WAKE_STALL_SECS=180 # minimum interval with no change of the oldest actionable foreign wake-queue row (it advances as the mate drains, and a queue reprovisioned under the same task id starts a fresh interval at whatever sequence it restarts) before an endpoint-recorded local secondmate produces one durable parent wake-loop-stall notification for that no-progress episode; a mate that is provably inside an active turn (an exact busy verdict) does not escalate until that same no-progress interval reaches FM_BUSY_TURN_MAX_SECS above; a mate whose busy class is exactly idle, whose agent is alive, and whose composer is not pending is rung once so its own home can drain, and the parent notification is withheld until that same row stays frozen for another stall interval; unknown or ring-unsafe panes keep the parent alarm; declared external-wait pause rows are excluded, and zero or invalid values use 180 -FM_SECONDMATE_LIVENESS_SECS=60 # seconds between watcher probes of each registered secondmate's recorded endpoint through bin/fm-secondmate-liveness-lib.sh, which relaunches only a positively `dead` or `missing` endpoint through the ordinary guarded fm-spawn.sh --secondmate path and emits exactly one check wake per relaunch; zero or invalid values use 60 -FM_SECONDMATE_LIVENESS_TIMEOUT=120 # seconds bounding one watcher-driven relaunch, so a wedged spawn cannot stall the poll; zero or invalid values use 120 -FM_SECONDMATE_LIVENESS_MAX_ATTEMPTS=3 # automatic relaunch attempts allowed per mate inside the window before the watcher parks auto-relaunch behind state/.secondmate-relaunch-bound- and escalates once; a later live probe clears the marker and restores the full attempt budget (the ledger keeps its history behind a `rearmed` row); zero or invalid values use 3 -FM_SECONDMATE_LIVENESS_WINDOW_SECS=3600 # window the relaunch bound counts state/.secondmate-relaunch- attempt lines over; the file is also the durable per-mate relaunch record; zero or invalid values use 3600 FM_WEDGE_DEMAND_INSPECT_COUNT=3 # consecutive provably-working stale escalations on the same unchanged pane before demand-deep-inspection is added +FM_WEDGE_MAX_ESCALATIONS=0 # OPT-IN: consecutive wedge escalations on the same window before the watcher emits one terminal "PERMANENTLY-WEDGED" wake and writes BOTH STATE/.wedge-permanent- (window-scoped) and STATE/.wedge-permanent-- (per-hash); subsequent polls for ANY hash in that window short-circuit until FM_CAP_HORIZON_SECS elapse or the operator manually removes BOTH markers. DEFAULT 0 = CAP DISABLED (every wedge escalation produces a wake, pre-PR behavior). Set FM_WEDGE_MAX_ESCALATIONS=N (N>=1) to opt in to the cap; N=10 caps after roughly 10 * STALE_ESCALATE_SECS (default 240s) = ~40 minutes of unattended signaling. Rejected values (non-integer) fall back to 0 with a triage_log warning; 0 is VALID (cap disabled, the unconfigured path). v20 (2026-09-24) flip: default disabled, opt-in by setting N>=1 — addresses the VISION.md "Authority is explicit" concern kunchenguid's firstmate raised on PR #2605 (unconfigured path no longer changes behavior for every captain). +FM_CAP_HORIZON_SECS=86400 # seconds a wedge cap marker is honored; a wedge older than this on the same window can re-fire. Bounds the silent-suppression window without depending on pause_state_class=working (which can be a steady state during a wedge, not a recovery signal); rejected values (0, non-integer) fall back to 86400 with a triage_log warning +FM_ROLLBACK_SENTINEL_TTL_SECS=3600 # seconds the .wedge-rollback-failed- sentinel short-circuits the wedge path after a cap-marker write failure AND its rollback also failed (same fs condition broke both). While fresh, the watcher emits no further PERMANENTLY-WEDGED wakes - one observed failure per wedge-event instead of a queue-flood. The sentinel expires naturally so a transient fs blip does not permanently silence the wedge; operator can `rm` it for immediate re-engagement. Rejected values (0, non-integer) fall back to 3600 with a triage_log warning FM_WORKTREE_WRITE_PRUNE='.git node_modules .venv venv __pycache__ .mypy_cache .pytest_cache .ruff_cache .tox target dist build .next .cache vendor' # directory names the wedge detector's task-worktree write probe skips; the default keeps .git out so a supervisor's own read-only git command can never look like crew progress; set it to the empty string to prune nothing, which widens the probe to the whole depth-bounded tree rather than disabling it FM_WORKTREE_WRITE_MAXDEPTH=6 # depth that same probe walks below the recorded worktree; it runs only at the moment a wedge escalation would otherwise fire, never on every poll; no probe knob applies to a secondmate, whose recorded worktree is a provisioned home the probe skips entirely FM_WORKTREE_WRITE_TIMEOUT=10 # wall-clock seconds that one walk may take, so a worktree on a hung mount cannot stall the watcher poll that started it; hitting the bound reads as no write evidence, which leaves the escalation schedule exactly as it was; a value that is not a positive integer falls back to the default @@ -2356,11 +1253,6 @@ FM_CRASH_BACKOFF=60 # seconds to wait after crossing the crash th FM_CRASH_NORMAL_SLEEP=5 # seconds to wait after an isolated watcher crash FM_LOG_MAX_BYTES=1048576 # daemon log size that triggers trimming FM_LOG_KEEP_LINES=2000 # daemon log lines kept when trimming -# supervision host (bin/fm-supervision-host.sh); read only in a home with config/supervision-host -FM_SUPERVISION_HOST_PARK_SECONDS=27000 # the host ends its park with a cycle-boundary wake after this long, under the Stop hook's 28800 s timeout -FM_SUPERVISION_HOST_TURN_TIMEOUT=1200 # bound on one engine turn; a turn that hits it hands its wake to main -FM_SUPERVISION_HOST_ROTATE_TURNS=20 # the engine conversation starts fresh after this many turns (and at every main session start) -FM_SUPERVISION_ENGINE_GRACE=30 # seconds between TERM and KILL when an engine turn is stopped # spoken interface and captain inbox; see "Spoken interface and captain inbox" above FM_VOICE_REGION= # overrides config/voice-region for one relay run FM_VOICE_MODEL= # overrides config/voice-model for one relay run @@ -2376,17 +1268,13 @@ FM_INBOX_PROFILE= # overrides config/inbox-profile; explicitly empty force `fm-teardown.sh` retries only Git's `Unable to create '...index.lock': File exists` return failure up to `FM_TREEHOUSE_RETURN_LOCK_RETRIES` times. `FM_TREEHOUSE_RETURN_LOCK_RETRIES` accepts a nonnegative integer, and an unset, blank, or invalid value uses the default of 3. - `FM_TREEHOUSE_RETURN_LOCK_RETRY_WAIT_SECS` accepts nonnegative whole or fractional seconds between attempts. When it is unset or blank, `FM_STALE_WORKTREE_LOCK_RETRY_WAIT_SECS` remains a compatible fallback, and a blank fallback uses the 1-second default. - An invalid nonblank wait falls back to 1 second rather than interrupting teardown. Teardown never removes a lock during the retry window, and after that window it attempts stale-lock cleanup only for a still-present lock that passes the configured age and live-holder checks. `fm-fleet-sync.sh` applies the same shape to an orphaned `.git/packed-refs.lock`: it retries only Git's `Unable to create '...packed-refs.lock': File exists` fetch failure up to `FM_FLEET_SYNC_PACKED_REFS_LOCK_RETRIES` times (nonnegative integer; unset, blank, or invalid uses the default of 3), waiting `FM_FLEET_SYNC_PACKED_REFS_LOCK_RETRY_WAIT_SECS` seconds (nonnegative whole or fractional; invalid falls back to 1 second) before each. Only after those retries exhaust does it remove the lock, and only when it is provably stale - still present, mtime age at least `FM_FLEET_SYNC_PACKED_REFS_LOCK_AGE_SECS` (default 30), and no `lsof` holder of the lock file or of the clone worktree itself (a live `git` keeps that as its cwd even in the window after it closes the lock and before it exits). - A live lock, a missing `lsof`, any failed check, or any other fetch failure keeps today's behavior. Every wait, retry, and removal is printed to stderr, and a successful recovery also prints one `recovered:` summary line to stdout so a session-start refresh - which discards fleet-sync stderr and relays only stdout - still surfaces it. - The shared staleness proof lives in `bin/fm-lock-lib.sh`, which both `fm-teardown.sh` and `fm-fleet-sync.sh` use. diff --git a/tests/fm-watch-wedge-cap.test.sh b/tests/fm-watch-wedge-cap.test.sh new file mode 100644 index 00000000000..98080e6f110 --- /dev/null +++ b/tests/fm-watch-wedge-cap.test.sh @@ -0,0 +1,1429 @@ +#!/usr/bin/env bash +# tests/fm-watch-wedge-cap.test.sh - focused unit tests for the +# FM_WEDGE_MAX_ESCALATIONS cap (local patch 2026-08-19, v6). Verifies: +# 1. cap fires PERMANENTLY-WEDGED at the threshold and writes BOTH +# markers: STATE/.wedge-permanent- (window-scoped) and +# STATE/.wedge-permanent-- (per-hash); +# 2. subsequent polls for any hash in that window are silent (no extra +# wakes) - fresh hashes included, per the v12 window-scoped gate; +# 3. the cap persists across pause-class transitions (paused: then +# lifted) - Greptile R4 fix; +# 4. the cap is bound by FM_CAP_HORIZON_SECS (re-fires after the +# horizon elapses) - Greptile R8 fix (the exit conditions are the +# horizon and an operator removing BOTH markers; a pane hash change +# alone does not re-engage while the window-scoped marker stands); +# 5. invalid override values (0, non-integer) fall back to the default +# for FM_WEDGE_MAX_ESCALATIONS and FM_CAP_HORIZON_SECS. +set -u + +# shellcheck source=tests/wake-helpers.sh +. "$(dirname "${BASH_SOURCE[0]}")/wake-helpers.sh" +# shellcheck source=/dev/null +. "$ROOT/bin/fm-classify-lib.sh" + +WATCH="$ROOT/bin/fm-watch.sh" +DRAIN="$ROOT/bin/fm-wake-drain.sh" +TMP_ROOT=$(fm_test_tmproot fm-watch-wedge-cap-tests) + +ack_stopped_cycle() { # + local state=$1 err sequence generation + err="$state/.test-cycle-drain.err" + FM_STATE_OVERRIDE="$state" "$DRAIN" >/dev/null 2> "$err" || return 1 + sequence=$(sed -n 's/^WAKE_ACK_REQUIRED:.*--ack-through \([0-9][0-9]*\) --recovery-generation [A-Za-z0-9._-][A-Za-z0-9._-]*$/\1/p' "$err") + generation=$(sed -n 's/^WAKE_ACK_REQUIRED:.*--ack-through [0-9][0-9]* --recovery-generation \([A-Za-z0-9._-][A-Za-z0-9._-]*\)$/\1/p' "$err") + rm -f "$err" + [ -n "$sequence" ] && [ -n "$generation" ] || return 1 + FM_STATE_OVERRIDE="$state" "$DRAIN" --ack-through "$sequence" \ + --recovery-generation "$generation" +} + +reap() { kill "$1" 2>/dev/null || true; wait "$1" 2>/dev/null || true; } + +is_live_non_zombie() { + local pid=$1 stat + kill -0 "$pid" 2>/dev/null || return 1 + stat=$(ps -p "$pid" -o stat= 2>/dev/null || true) + case "$stat" in + Z*) return 1 ;; + esac + return 0 +} + +wait_for_exit() { + local pid=$1 limit=${2:-50} i=0 + while [ "$i" -lt "$limit" ]; do + if ! is_live_non_zombie "$pid"; then + wait "$pid" + return "$?" + fi + sleep 0.1 + i=$((i + 1)) + done + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + return 124 +} + +file_mtime() { + if [ "$(uname)" = Darwin ]; then stat -f %m "$1" 2>/dev/null; else stat -c %Y "$1" 2>/dev/null; fi +} + +wait_poll_cycle() { # [limit-ticks] + local state=$1 pid=$2 limit=${3:-300} beat first now i=0 + beat="$state/.last-watcher-beat" + rm -f "$beat" + first="" + while [ "$i" -lt "$limit" ]; do + kill -0 "$pid" 2>/dev/null || return 1 + first=$(file_mtime "$beat") + [ -n "$first" ] && break + sleep 0.1 + i=$((i + 1)) + done + while [ "$i" -lt "$limit" ]; do + kill -0 "$pid" 2>/dev/null || return 1 + now=$(file_mtime "$beat") + if [ -n "$now" ] && [ "$now" != "$first" ]; then + return 0 + fi + sleep 0.1 + i=$((i + 1)) + done + return 1 +} + +seen_sig() { + if [ "$(uname)" = Darwin ]; then stat -f '%z:%Fm' "$1" 2>/dev/null; else stat -c '%s:%Y' "$1" 2>/dev/null; fi +} + +# --- FM_WEDGE_MAX_ESCALATIONS cap (local patch 2026-08-19, v6) ---------------- +# The cap is a hard floor on the LLM-supervised unattended loop that the 2026- +# 08-18 MiniMax drain (~359M tokens) demonstrated. Past FM_WEDGE_MAX_ESCALATIONS +# consecutive wedge escalations on the same window, the watcher emits ONE +# terminal wake with PERMANENTLY-WEDGED and writes BOTH STATE/.wedge-permanent- +# - (per-hash) and STATE/.wedge-permanent- (window-scoped, +# v12); each marker's content is the cap-fire epoch. Two exit conditions: +# FM_CAP_HORIZON_SECS elapses since that timestamp (default 24h), or the +# operator manually removes BOTH markers - a pane hash change alone does NOT +# re-engage while the window-scoped marker stands (v12). No auto-lift on +# pause_state_class=working (the v6/v7 lift sites were removed in v9 because +# pause_state_class=working can be a steady state during a wedge, not a recovery +# signal). + +test_wedge_cap_window_marker_silences_hash_churning_busy_pane() { + # v12 regression for Greptile P1 follow-up: a worker that churns its rendered + # pane hash on every poll (a ticking elapsed-time footer, a pane re-rendering + # for any unrelated reason) must receive at most ONE terminal PERMANENTLY- + # WEDGED wake for the wedge-event, even when every poll presents a fresh + # hash that escapes the per-hash marker scheme. v12 introduces a window- + # scoped marker (in addition to the per-hash marker) so the wedge is silenced + # for the whole window until the cap horizon elapses or an operator rm + # clears the marker. + local dir state fakebin out capture_file window key pane_hash sig pid n max marker_window marker_hash + dir=$(make_case wedge-cap-churning); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-churning" + printf 'busy wedged pane initial content\n' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/wedge-cap-churning.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-churning.status" + sig=$(seen_sig "$state/wedge-cap-churning.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-churning_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "busy wedged pane initial content") + printf '%s' "$pane_hash" > "$state/.hash-$key" + # v17 (2026-09-11): churn test now exercises busy_turn_bound_check → + # wedge_timer_check instead of the idle hash-change branch, so the + # window-scoped marker is actually checked against a fresh hash per + # poll. The busy verdict comes from the production semantic busy-state + # contract (harness=pi + an armed .busy-state record written by the real + # fm-busy-event.sh writer - a bare crew-state string is NOT trusted by + # window_is_busy), FM_BUSY_TURN_MAX_SECS=1, .meta aged past 1s. + printf 'busy: harness busy\n' > "$state/wedge-cap-churning.status" + sig=$(seen_sig "$state/wedge-cap-churning.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-churning_status" + touch -d '2 seconds ago' "$state/wedge-cap-churning.meta" 2>/dev/null || \ + perl -e 'utime(time()-2, time()-2, $ARGV[0])' "$state/wedge-cap-churning.meta" + printf '1\n' > "$state/.count-$key" + max=3 + export FM_FAKE_CREW_STATE='state: working · source: pane · harness busy' + "$ROOT/bin/fm-busy-event.sh" arm "$state" "wedge-cap-churning" --state busy --source pi-ext --event poll >/dev/null \ + || fail "could not arm the busy-state record for the churn fixture" + marker_window="$state/.wedge-permanent-$key" + marker_hash="$state/.wedge-permanent-$key-${pane_hash:0:12}" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 FM_BUSY_TURN_MAX_SECS=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "priming watch failed: $(cat "$out")" + fi + reap "$pid" + # The priming poll takes the busy route below the busy-turn bound, where + # wedge_timer_check only resets a missing timer - nothing actionable is + # queued, so there may be nothing to ack. + ack_stopped_cycle "$state" || true + n=1 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + touch -d '2 seconds ago' "$state/wedge-cap-churning.meta" 2>/dev/null || \ + perl -e 'utime(time()-2, time()-2, $ARGV[0])' "$state/wedge-cap-churning.meta" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 FM_BUSY_TURN_MAX_SECS=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_for_exit "$pid" 100; then + reap "$pid"; fail "round $n watch failed: $(cat "$out")" + fi + ack_stopped_cycle "$state" || fail "round $n ack failed" + n=$((n + 1)) + done + [ -e "$marker_hash" ] || fail "per-hash cap marker missing after firing" + [ -e "$marker_window" ] || fail "window-scoped cap marker missing after firing - v12 window-scope write is broken" + + total_terminal=0 + i=0 + while [ "$i" -lt 12 ]; do + i=$((i + 1)) + printf 'busy wedged pane iteration %d with brand new content\n' "$i" > "$capture_file" + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + touch -d '2 seconds ago' "$state/wedge-cap-churning.meta" 2>/dev/null || \ + perl -e 'utime(time()-2, time()-2, $ARGV[0])' "$state/wedge-cap-churning.meta" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=1 FM_BUSY_TURN_MAX_SECS=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max FM_CAP_HORIZON_SECS=86400 "$WATCH" > "$out" & + pid=$! + if wait_poll_cycle "$state" "$pid" 2>/dev/null; then + : + fi + reap "$pid" + ack_stopped_cycle "$state" || true + if grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null; then + total_terminal=$((total_terminal + 1)) + fi + done + [ "$total_terminal" -eq 0 ] || fail "hash-churning pane fired $total_terminal additional PERMANENTLY-WEDGED wakes across 12 iterations - v12 window-scoped marker is not silencing the loop" + [ -e "$marker_window" ] || fail "window-scoped marker was unexpectedly cleared during hash churn" + unset FM_FAKE_CREW_STATE + pass "the window-scoped cap marker silences hash-churning panes for the full cap horizon" +} + + +test_wedge_cap_marker_write_after_durable_wake_v13() { + # v13 (2026-09-08): marker writes happen AFTER the durable wake row is + # queued. This closes Greptile P1 #1 ("marker precedes durable wake"): + # if the watcher dies between the marker write and fm_wake_append, the + # marker silences retries even though no wake was queued. + # + # Test: drive the cap to fire with FM_WAKE_QUEUE pointing at a NON- + # writable path. fm_wake_append returns failure. Watcher exits 1 with + # NO cap marker written (per-hash AND window-scoped) - next poll must + # be able to re-escalate the wedge from scratch. + local dir state fakebin out capture_file window key pane_hash sig pid n max marker_hash marker_window + dir=$(make_case wedge-cap-write-order); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-write-order" + printf 'idle wedged content for v13 ordering' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-write-order.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-write-order.status" + sig=$(seen_sig "$state/wedge-cap-write-order.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-write-order_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle wedged content for v13 ordering") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=3 + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + marker_hash="$state/.wedge-permanent-$key-${pane_hash:0:12}" + marker_window="$state/.wedge-permanent-$key" + + # Allow the queue to be writable for rounds 1..max-1 so the normal stale + # escalations land in the durable queue. On the firing round (round + # max), we lock the queue directory so fm_wake_append will fail. v13 + # asserts the cap path does NOT write any marker when fm_wake_append + # fails - this is the only way to exercise the v13 ordering without + # also breaking normal stale wakes. + mkdir -p "$dir/queue-parent" + printf '%s\n' "# prior-round wakes" > "$dir/queue-parent/queue" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "priming watch failed: $(cat "$out")" + fi + reap "$pid" + ack_stopped_cycle "$state" || fail "priming ack failed" + n=1 + while [ "$n" -le "$max" ]; do + if [ "$n" -eq "$max" ]; then + # Make fm_wake_append (the LAST step before the per-hash + + # window-scoped marker writes in v12) fail on the firing round: + # remove the queue file and lock the queue directory, so the + # append's create cannot succeed (appending to the pre-existing + # writable file would succeed even under a read-only parent). The + # v13 fix asserts: NO marker is written when fm_wake_append fails. + rm -f "$dir/queue-parent/queue" + chmod 0555 "$dir/queue-parent" + fi + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max \ + FM_WAKE_QUEUE="$dir/queue-parent/queue" "$WATCH" > "$out" & + pid=$! + set +e + wait "$pid" 2>/dev/null + round_status=$? + if [ "$n" -eq "$max" ]; then + # Round max is expected to exit 1: v13 writes no marker when + # fm_wake_append fails and exits 1 so the next poll retries the + # cap from scratch. Other rounds must exit 0 or the watcher + # logic regressed. + [ "$round_status" -eq 1 ] || fail "round max expected exit 1 (v13 no-marker append-failure path), got $round_status: $(cat "$out")" + else + [ "$round_status" -eq 0 ] || fail "round $n watch failed with exit $round_status: $(cat "$out")" + fi + if [ "$n" -lt "$max" ]; then + ack_stopped_cycle "$state" || fail "round $n ack failed" + fi + n=$((n + 1)) + done + chmod 0755 "$dir/queue-parent" + [ ! -e "$marker_hash" ] || fail "per-hash marker was written despite fm_wake_append failure - v13 ordering is broken (the marker precedes the durable wake)" + [ ! -e "$marker_window" ] || fail "window-scoped marker was written despite fm_wake_append failure - v13 ordering is broken" + unset FM_FAKE_CREW_STATE + pass "cap-fire ordering durably queues the wake BEFORE any marker is written (v13)" +} + +test_wedge_cap_failed_window_marker_rolls_back_per_hash_v13() { + # v13 (2026-09-08): closes Greptile P1 #3 ("failed window marker permits + # repeats") and P1 #2 ("window marker survives append failure"). On a + # window-scoped marker write FAILURE (after fm_wake_append success and + # per-hash marker success), v13 rolls back the per-hash marker and + # exits 1 - leaving no partial cap state visible. + # + # Test: pre-create $state/.wedge-permanent- as a NON-EMPTY DIRECTORY + # (so `date +%s > ...` cannot create a file at that exact path - bash + # refuses to truncate a directory). The v13 fix rolls back the per-hash + # marker (which is targeted at .wedge-permanent--, a + # different file) when the window-scoped marker write fails. We assert + # the rollback fired: neither marker ends up on disk. + local dir state fakebin out capture_file window key pane_hash sig pid n max marker_hash marker_window + dir=$(make_case wedge-cap-window-fail); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-window-fail" + printf 'idle wedged content for v13 window-fail' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-window-fail.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-window-fail.status" + sig=$(seen_sig "$state/wedge-cap-window-fail.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-window-fail_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle wedged content for v13 window-fail") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=3 + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + marker_hash="$state/.wedge-permanent-$key-${pane_hash:0:12}" + marker_window="$state/.wedge-permanent-$key" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "priming watch failed: $(cat "$out")" + fi + reap "$pid" + ack_stopped_cycle "$state" || fail "priming ack failed" + n=1 + while [ "$n" -le "$max" ]; do + if [ "$n" -eq "$max" ]; then + # On the firing round, plant the window-scoped marker path as an + # EXISTING DIRECTORY so `date +%s > "$STATE/.wedge-permanent-"` + # cannot create the file (bash refuses to redirect into a directory). + # v13 then rolls back the per-hash marker (which IS writable). + mkdir -p "$marker_window/blocker" 2>/dev/null + fi + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + set +e + wait "$pid" 2>/dev/null + round_status=$? + if [ "$n" -eq "$max" ]; then + # Round max is expected to exit 1: v13 rolls back the per-hash + # marker after a failed window-scoped marker write and exits 1. + # Other rounds must exit 0 or the watcher logic regressed. + [ "$round_status" -eq 1 ] || fail "round max expected exit 1 (v13 cap-failure rollback), got $round_status: $(cat "$out")" + else + [ "$round_status" -eq 0 ] || fail "round $n watch failed with exit $round_status: $(cat "$out")" + fi + if [ "$n" -lt "$max" ]; then + ack_stopped_cycle "$state" || fail "round $n ack failed" + fi + n=$((n + 1)) + done + rm -rf "$marker_window" 2>/dev/null || true + [ ! -e "$marker_hash" ] || fail "per-hash marker was NOT rolled back when the window-scoped marker write failed - v13 rollback is incomplete" + [ ! -e "$marker_window" ] || fail "window-scoped marker is unexpectedly present after a failed write - v13 ordering created an inconsistency" + unset FM_FAKE_CREW_STATE + pass "v13 rolls back the per-hash marker when the window-scoped marker write fails (no partial cap state)" +} + + +test_wedge_cap_rollback_resets_state_v14() { + # v14 (2026-09-08): closes Greptile P1 from v13 review - "Failed cap + # writes refire immediately". On either cap-marker write failure the + # watcher must: + # - reset .wedge-escalations- (so the next poll starts at 1, not + # at the saturated value) + # - reset .stale-since- to "now" (so the next poll ages from 0, + # not 500s ago) + # - clear_write_tracking on the window key + # Otherwise the next poll (which runs in a fresh watcher invocation) reads + # the saturated n and the stale timer, fires PERMANENTLY-WEDGED again + # immediately, and repeats every poll while the write failure persists. + # + # Test: drive the cap with FM_WAKE_QUEUE writable (so fm_wake_append + # succeeds) and with the window-scoped marker path planted as a directory + # (so the per-hash marker write succeeds and the window-scoped marker + # write fails - the v13 failure path). After round max we inspect the + # post-rollback state: escalation counter is 0, stale timer is "recent" + # (within last 5 seconds), no per-hash marker. + local dir state fakebin out capture_file window key pane_hash sig pid n max marker_hash marker_window + dir=$(make_case wedge-cap-rollback); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-rollback" + printf 'idle wedged content for v14 rollback' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-rollback.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-rollback.status" + sig=$(seen_sig "$state/wedge-cap-rollback.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-rollback_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle wedged content for v14 rollback") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=3 + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + marker_hash="$state/.wedge-permanent-$key-${pane_hash:0:12}" + marker_window="$state/.wedge-permanent-$key" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "priming watch failed: $(cat "$out")" + fi + reap "$pid" + ack_stopped_cycle "$state" || fail "priming ack failed" + n=1 + while [ "$n" -le "$max" ]; do + if [ "$n" -eq "$max" ]; then + # Plant the window-scoped marker path as a non-empty directory so + # the per-hash marker write succeeds (closing P1 #2/#3 in v13) and + # the window-scoped marker write fails - triggering the v14 rollback. + mkdir -p "$marker_window/blocker" 2>/dev/null + fi + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + set +e + wait "$pid" 2>/dev/null + round_status=$? + if [ "$n" -eq "$max" ]; then + [ "$round_status" -eq 1 ] || fail "round max expected exit 1 (v14 cap-failure rollback), got $round_status: $(cat "$out")" + else + [ "$round_status" -eq 0 ] || fail "round $n watch failed with exit $round_status: $(cat "$out")" + fi + if [ "$n" -lt "$max" ]; then + ack_stopped_cycle "$state" || fail "round $n ack failed" + fi + n=$((n + 1)) + done + + # v14 assertions: after the cap-failure rollback: + # 1. .wedge-permanent-- must NOT exist (per-hash rolled back) + # 2. .wedge-escalations- must be 0 (counter rolled back to 0) + # 3. .stale-since- must be "recent" (within 5 seconds of now; + # rolled back so next poll ages from "now") + rm -rf "$marker_window" 2>/dev/null || true + [ ! -e "$marker_hash" ] || fail "per-hash marker was NOT rolled back - v14 rollback is incomplete (P1 #2/#3 regressed)" + if [ -e "$state/.wedge-escalations-$key" ]; then + ewf_after=$(cat "$state/.wedge-escalations-$key" 2>/dev/null || echo "") + case "$ewf_after" in + ''|"0") ;; + *) fail "wedge-escalation counter was NOT reset by v14 rollback (still holds '$ewf_after' = saturation value) - the next poll will re-fire PERMANENTLY-WEDGED immediately (this is the P1 Greptile flagged on v13)" + ;; + esac + fi + if [ -e "$state/.stale-since-$key" ]; then + ssf_age=$(($(date +%s) - $(cat "$state/.stale-since-$key"))) + [ "$ssf_age" -le 5 ] || fail "stale-since timer was NOT refreshed by v14 rollback (now ${ssf_age}s old - .stale-since- still ages from 500s ago at the next poll, which means the next poll re-fires PERMANENTLY-WEDGED immediately)" + else + fail "stale-since was removed entirely (v14 rollback should RESTORE it to 'now', not delete it - the wedge timer needs an anchor)" + fi + unset FM_FAKE_CREW_STATE + pass "v14 rollback resets both the escalation counter and the stale timer so the next poll re-escalates from 1, not the saturated value (closes Greptile P1 from v13 review)" +} + + +test_wedge_cap_rollback_failure_sets_sentinel_v15() { + # v15 (2026-09-08): closes Greptile P1 from v14 review - "Rollback + # failures preserve saturation". v14 used `|| true` on every rollback + # line, so a rollback that itself failed (the SAME fs condition that + # broke the marker write) would silently preserve the saturation. v15: + # + # 1. _wedge_cap_rollback returns 1 if any reset fails AND writes a + # .wedge-rollback-failed- sentinel (timestamp + first failing + # path). + # 2. wedge_timer_check checks for the sentinel at the top - if recent + # (within FM_ROLLBACK_SENTINEL_TTL_SECS, default 3600s), it returns 0 + # without publishing a wake or writing a marker. + # + # This test exercises the top-of-function sentinel check directly: + # plant a sentinel file with a recent timestamp, drive a poll, and + # confirm the wedge path short-circuits (no PERMANENTLY-WEDGED wake). + # (The cap path's exit-2 path is exercised indirectly by ensuring the + # sentinel state in STATE survives a round; see also + # test_wedge_cap_rollback_resets_state_v14 for the rollback SUCCESS + # path. Together they cover the v15 contract.) + local dir state fakebin out capture_file window key pane_hash sig pid n max marker_hash marker_window sentinel + dir=$(make_case wedge-cap-rollback-fails); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-rollback-fails" + printf 'idle wedged content for v15 rollback-fails' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-rollback-fails.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-rollback-fails.status" + sig=$(seen_sig "$state/wedge-cap-rollback-fails.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-rollback-fails_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle wedged content for v15 rollback-fails") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=3 + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + marker_hash="$state/.wedge-permanent-$key-${pane_hash:0:12}" + marker_window="$state/.wedge-permanent-$key" + sentinel="$state/.wedge-rollback-failed-$key" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "priming watch failed: $(cat "$out")" + fi + reap "$pid" + ack_stopped_cycle "$state" || fail "priming ack failed" + n=1 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 100 || { reap "$pid"; fail "round $n watch failed: $(cat "$out")"; } + ack_stopped_cycle "$state" || fail "round $n ack failed" + n=$((n + 1)) + done + [ -e "$marker_hash" ] || fail "per-hash cap marker missing after firing - prime test before sentinel test must pass first" + [ -e "$marker_window" ] || fail "window-scoped cap marker missing after firing - prime test before sentinel test must pass first" + + # Now plant the v15 sentinel: .wedge-rollback-failed- with a recent + # timestamp. Drive ONE more poll and confirm the wedge path short-circuits. + # The short-circuit is by design a non-exiting path: the watcher stays + # alive (it's still polling the window) but wedge_timer_check returns 0 + # at the top without publishing a wake. The test uses wait_poll_cycle + # (which waits for a heartbeat) to confirm the watcher is alive AND + # processing the sentinel without publishing a wake. + # + # triage_log writes to STATE/.watch-triage.log, not stdout - the test + # checks that file (not $out) for the expected short-circuit log line. + printf '%s %s\n' "$(date +%s)" "test-injected" > "$sentinel" + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max FM_ROLLBACK_SENTINEL_TTL_SECS=3600 "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "v15 sentinel poll failed (watcher never beat): $(cat "$out")" + fi + if grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null; then + reap "$pid"; fail "v15 sentinel short-circuit published a PERMANENTLY-WEDGED wake - the queue-flood behavior v15 closed is broken" + fi + if ! grep -F "rollback-failed sentinel active" "$state/.watch-triage.log" >/dev/null; then + reap "$pid"; fail "expected triage_log 'rollback-failed sentinel active' was not emitted (triage file: $(cat "$state/.watch-triage.log" 2>/dev/null || echo missing))" + fi + [ -e "$sentinel" ] || { reap "$pid"; fail "v15 sentinel was removed by the short-circuit poll - it should remain until TTL expiry or operator rm"; } + reap "$pid" + + # Final cleanup of the primed cap markers so the test exits clean. + rm -f "$marker_hash" "$marker_window" + + unset FM_FAKE_CREW_STATE + pass "v15 rollback-failed sentinel short-circuits the wedge path - no wake-amplification under persistent fs failure (closes Greptile P1 from v14 review)" +} + + +test_wedge_cap_rollback_sentinel_keyed_on_busy_route_v17() { + # v17 (2026-09-11): busy-pane regression for the rollback-failed sentinel. + # busy_turn_bound_check declares an empty `local key` on its fall-through + # path, and bash dynamic scoping used to shadow wedge_timer_check's key + # with that empty value when the cap path handed off to _wedge_cap_rollback + # - so on the busy route the sentinel was written as + # .wedge-rollback-failed- (empty key) while the next poll's keyed lookup + # reads .wedge-rollback-failed-: the v15 short-circuit silently did + # not hold exactly where v17's fix targets it. Drive the REAL busy route + # (busy crew state + crossed busy-turn bound) through a failing + # window-marker write whose rollback reset also fails, then require the + # NEXT busy-route poll to short-circuit on the keyed sentinel. + local dir state fakebin out capture_file window key pane_hash sig pid max + local marker_window marker_hash sentinel esc esc_after round_status rollbacks + dir=$(make_case wedge-cap-busy-route-sentinel); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-busy-sentinel" + printf 'busy wedged pane for v17 busy-route sentinel\n' > "$capture_file" + printf 'window=%s\nkind=ship\nharness=pi\n' "$window" > "$state/fm-wedge-cap-busy-sentinel.meta" + printf 'busy: harness busy\n' > "$state/fm-wedge-cap-busy-sentinel.status" + sig=$(seen_sig "$state/fm-wedge-cap-busy-sentinel.status"); printf '%s' "$sig" > "$state/.seen-fm-wedge-cap-busy-sentinel_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "busy wedged pane for v17 busy-route sentinel") + printf '%s' "$pane_hash" > "$state/.hash-$key" + max=1 + export FM_FAKE_CREW_STATE='state: working · source: pane · harness busy' + # The busy verdict comes from the production semantic busy-state contract: + # harness=pi in the meta plus an armed .busy-state record written by the + # real fm-busy-event.sh writer. A bare crew-state string is NOT trusted by + # window_is_busy, so without this the poll silently takes an idle absorb + # path and never reaches busy_turn_bound_check. + "$ROOT/bin/fm-busy-event.sh" arm "$state" "fm-wedge-cap-busy-sentinel" --state busy --source pi-ext --event poll >/dev/null \ + || fail "could not arm the busy-state record for the busy-route fixture" + marker_window="$state/.wedge-permanent-$key" + marker_hash="$state/.wedge-permanent-$key-${pane_hash:0:12}" + sentinel="$state/.wedge-rollback-failed-$key" + esc="$state/.wedge-escalations-$key" + + # Round 1: the cap fires on the busy route (max=1, fresh escalation), the + # window-scoped marker write fails (non-empty directory blocker), and the + # rollback's escalation-file reset ALSO fails (the escalation path is the + # same kind of blocker) - so the rollback must write the keyed sentinel + # and the cap path must exit 2. + mkdir -p "$esc/blocker" + mkdir -p "$marker_window/blocker" + touch -d '2 seconds ago' "$state/fm-wedge-cap-busy-sentinel.meta" 2>/dev/null || \ + perl -e 'utime(time()-2, time()-2, $ARGV[0])' "$state/fm-wedge-cap-busy-sentinel.meta" + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 FM_BUSY_TURN_MAX_SECS=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max FM_ROLLBACK_SENTINEL_TTL_SECS=3600 "$WATCH" > "$out" & + pid=$! + round_status=0 + wait_for_exit "$pid" 100 || round_status=$? + [ "$round_status" -ne 124 ] || fail "round 1 watch did not exit after the cap-failure rollback: $(cat "$out")" + [ -e "$sentinel" ] || { rollbacks=''; for f in "$state"/.wedge-rollback-failed*; do if [ -e "$f" ]; then rollbacks="$rollbacks $f"; fi; done; fail "round 1 wrote no sentinel at .wedge-rollback-failed-$key - the busy route built the sentinel name from an unscoped key (pre-v17 shadowing) or the rollback sentinel write is broken (found:${rollbacks:- none})"; } + [ "$round_status" -eq 2 ] || fail "round 1 expected exit 2 (rollback-failed sentinel set), got $round_status: $(cat "$out")" + ack_stopped_cycle "$state" || fail "round 1 ack failed" + + # Round 2: restore a healthy fs (blockers gone, per-hash and window + # markers removed, timer re-aged, counter reset to 0) and drive one more + # busy-route poll. The keyed sentinel must short-circuit it: no + # PERMANENTLY-WEDGED wake, the sentinel-active triage line, the sentinel + # retained, and the escalation counter untouched. Pre-v17 the keyed + # lookup missed and this poll re-fired the cap - the queue-flood the + # sentinel exists to stop. + rm -rf "$marker_window" "$esc" + rm -f "$marker_hash" + printf '0\n' > "$esc" + touch -d '2 seconds ago' "$state/fm-wedge-cap-busy-sentinel.meta" 2>/dev/null || \ + perl -e 'utime(time()-2, time()-2, $ARGV[0])' "$state/fm-wedge-cap-busy-sentinel.meta" + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + rm -f "$state/.watch-triage.log" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 FM_BUSY_TURN_MAX_SECS=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max FM_ROLLBACK_SENTINEL_TTL_SECS=3600 "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + : # a pre-fix regression exits on its own after re-firing; assertions below catch it + fi + reap "$pid" + ack_stopped_cycle "$state" || true + if grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null; then + fail "busy-route poll re-fired PERMANENTLY-WEDGED despite the keyed rollback sentinel - the v15 short-circuit does not hold on the busy-pane route (pre-v17 empty-key shadowing): $(cat "$out")" + fi + if ! grep -F "rollback-failed sentinel active" "$state/.watch-triage.log" >/dev/null; then + fail "expected triage_log 'rollback-failed sentinel active' was not emitted on the busy route (triage file: $(cat "$state/.watch-triage.log" 2>/dev/null || echo missing))" + fi + [ -e "$sentinel" ] || fail "the short-circuit poll removed the keyed sentinel - it must remain until TTL expiry or operator rm" + esc_after=$(cat "$esc" 2>/dev/null || true) + [ "$esc_after" = "0" ] || fail "the sentinel short-circuit poll escalated anyway (counter now '$esc_after', expected untouched 0)" + unset FM_FAKE_CREW_STATE + pass "the rollback-failed sentinel is written and honored under its proper window key on the busy-pane route (closes the v17 dynamic-scoping shadowing)" +} + +test_wedge_cap_fires_permanently_wedged_after_max_escalations() { + local dir state fakebin out capture_file window key pane_hash sig pid n max + dir=$(make_case wedge-cap-fires); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap" + printf 'idle wedged content' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap.meta" + printf 'working: still wedged\n' > "$state/wedge-cap.status" + sig=$(seen_sig "$state/wedge-cap.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle wedged content") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=4 + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + + # Priming round: pre-seeded .hash and .count mean one wait_poll_cycle + # reaches the wedge path (n=2 from the count pre-seed + increment). + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "watcher exited on the priming round (should absorb): $(cat "$out")" + fi + reap "$pid" + ack_stopped_cycle "$state" || fail "could not acknowledge the priming stop" + + # Drive past the cap (max=4). Rounds 1..3 are normal escalations; round 4 fires PERMANENTLY-WEDGED. + n=1 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_for_exit "$pid" 100; then + reap "$pid"; fail "watcher did not exit on wedge round $n: $(cat "$out")" + fi + grep -F "escalation $n" "$out" >/dev/null || fail "round $n did not report escalation count $n: $(cat "$out")" + if [ "$n" -lt "$max" ]; then + grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null && fail "round $n fired PERMANENTLY-WEDGED before the cap" + else + grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null || fail "round $max (cap) did not produce PERMANENTLY-WEDGED: $(cat "$out")" + fi + ack_stopped_cycle "$state" || fail "could not acknowledge wedge round $n" + n=$((n + 1)) + done + + # The per-(window, hash) marker must be set. + [ -e "$state/.wedge-permanent-$key-${pane_hash:0:12}" ] || fail "cap marker .wedge-permanent-- was not written after the cap fired" + unset FM_FAKE_CREW_STATE + pass "wedge cap fires PERMANENTLY-WEDGED at FM_WEDGE_MAX_ESCALATIONS and writes the per-hash marker" +} + +test_wedge_cap_suppresses_subsequent_polls_for_same_hash() { + local dir state fakebin out capture_file window key pane_hash sig pid n max + dir=$(make_case wedge-cap-suppress); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-suppress" + printf 'idle wedged content' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-suppress.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-suppress.status" + sig=$(seen_sig "$state/wedge-cap-suppress.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-suppress_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle wedged content") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=3 + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + + # Priming + cap-firing rounds, condensed. + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_poll_cycle "$state" "$pid" || { reap "$pid"; fail "priming watch failed"; } + reap "$pid" + ack_stopped_cycle "$state" || fail "priming ack failed" + + n=1 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 100 || { reap "$pid"; fail "round $n watch failed"; } + ack_stopped_cycle "$state" || fail "round $n ack failed" + n=$((n + 1)) + done + + # Cap marker must exist after the cap fired. + [ -e "$state/.wedge-permanent-$key-${pane_hash:0:12}" ] || fail "cap marker missing before the suppression check" + + # Now run a fresh watcher poll: pane is still wedged (same content, worker + # still NOT genuinely recovered - FM_FAKE_CREW_STATE=paused, so v7 site 3 does + # NOT lift the marker). The wedge_timer_check early-return path should fire. + # No wake should be queued. + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + # Worker is genuinely still wedged (paused state, not working) - this is an + # operator wait or stuck wedge, not a recovery. The cap must hold. + FM_FAKE_CREW_STATE='state: paused · source: run-step · waiting on external release' \ + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "watcher exited when the cap should have suppressed (the marker exists, watcher should have absorbed): $(cat "$out")" + fi + reap "$pid" + # The drain output must NOT contain a stale wake for this window - the cap + # short-circuited before fm_wake_append was reached. + drain_out="$dir/drain-after-suppress.out" + FM_STATE_OVERRIDE="$state" "$DRAIN" > "$drain_out" 2>/dev/null || true + if grep "$(printf '\tstale\t')" "$drain_out" 2>/dev/null | grep -F "$window" >/dev/null; then + fail "capped hash still produced a stale wake after the cap fired: $(cat "$drain_out")" + fi + unset FM_FAKE_CREW_STATE + pass "subsequent polls for the capped hash are silent - no additional terminal wakes fire" +} + +test_wedge_cap_persists_across_pause_class_transitions() { + local dir state fakebin out capture_file window key pane_hash sig pid max + dir=$(make_case wedge-cap-pause); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-pause" + printf 'idle wedged content' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-pause.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-pause.status" + sig=$(seen_sig "$state/wedge-cap-pause.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-pause_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle wedged content") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=3 + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + + # Drive to cap. + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_poll_cycle "$state" "$pid" || { reap "$pid"; fail "priming watch failed"; } + reap "$pid" + ack_stopped_cycle "$state" || fail "priming ack failed" + n=1 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 100 || { reap "$pid"; fail "round $n watch failed"; } + ack_stopped_cycle "$state" || fail "round $n ack failed" + n=$((n + 1)) + done + [ -e "$state/.wedge-permanent-$key-${pane_hash:0:12}" ] || fail "cap marker missing before pause-cycle test" + + # Operator declares paused: - but the worker is NOT actually working + # (FM_FAKE_CREW_STATE=paused), so this is an operator wait, not a recovery. + # pause_state_class returns "paused" (not "working"), so the v6 lift sites + # do NOT fire. The cap marker MUST persist (Greptile R4 fix). + printf 'paused: waiting on a human\n' > "$state/wedge-cap-pause.status" + sig=$(seen_sig "$state/wedge-cap-pause.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-pause_status" + printf 'idle wedged content' > "$capture_file" + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + # Worker is genuinely paused (not working) - this is an operator wait, not recovery. + FM_FAKE_CREW_STATE='state: paused · source: run-step · waiting on external release' \ + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + # In idle (no-actionable-wake) mode the watcher stays alive in its poll loop + # and the cap suppresses any escalation; a startup rearm-resurface check wake + # may or may not fire depending on the downtime-marker state from prior + # rounds. Wait for two distinct beat mtimes (one full poll cycle) to confirm + # the watcher has scanned the pane, then reap. Either way, the wedge_timer_check + # early-return on the cap marker means no stale wake is queued. + if ! wait_poll_cycle "$state" "$pid"; then + # Watcher may have exited via rearm-resurface check; drain and continue. + wait "$pid" 2>/dev/null || true + ack_stopped_cycle "$state" || true + fi + reap "$pid" + ack_stopped_cycle "$state" || true + [ -e "$state/.wedge-permanent-$key-${pane_hash:0:12}" ] || fail "cap marker was cleared after a paused: declaration with non-working crew (Greptile R4 regression)" + + # Operator lifts the pause, but the worker still isn't working - status returns + # to "working:" verb but FM_FAKE_CREW_STATE stays "paused" so pause_state_class + # returns "paused" (the status verb matches but the authoritative state says + # still waiting). Marker MUST persist. + printf 'working: back online\n' > "$state/wedge-cap-pause.status" + sig=$(seen_sig "$state/wedge-cap-pause.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-pause_status" + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + FM_FAKE_CREW_STATE='state: paused · source: run-step · waiting on external release' \ + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + wait "$pid" 2>/dev/null || true + ack_stopped_cycle "$state" || true + fi + reap "$pid" + ack_stopped_cycle "$state" || true + [ -e "$state/.wedge-permanent-$key-${pane_hash:0:12}" ] || fail "cap marker was cleared after the pause was lifted without an active pipeline (Greptile R4 regression)" + unset FM_FAKE_CREW_STATE + pass "the cap marker persists across pause: and unpause transitions when the worker is not actively recovered" +} + +# v9: cap is bound by FM_CAP_HORIZON_SECS, NOT by pause_state_class=working lift sites. +# v12: a new hash does NOT lift the cap - the window-scoped marker silences all +# hashes in the window; operator can `rm` BOTH markers for immediate +# re-engagement. See tests below for the new semantics. + +test_wedge_cap_expires_after_horizon() { + local dir state fakebin out capture_file window key pane_hash sig pid max marker + dir=$(make_case wedge-cap-horizon); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-horizon" + printf 'idle wedged content' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-horizon.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-horizon.status" + sig=$(seen_sig "$state/wedge-cap-horizon.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-horizon_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle wedged content") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=3 + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + marker="$state/.wedge-permanent-$key-${pane_hash:0:12}" + + # Drive to cap. + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_poll_cycle "$state" "$pid" || { reap "$pid"; fail "priming watch failed"; } + reap "$pid" + ack_stopped_cycle "$state" || fail "priming ack failed" + n=1 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 100 || { reap "$pid"; fail "round $n watch failed"; } + ack_stopped_cycle "$state" || fail "round $n ack failed" + n=$((n + 1)) + done + [ -e "$marker" ] || fail "cap marker missing before horizon test" + + # Backdate BOTH markers (per-hash AND window-scoped) so both appear older + # than FM_CAP_HORIZON_SECS. The cap is now stale on both gates; the next + # wedge_timer_check call must NOT be short-circuited by either marker. + old_ts=$(( $(date +%s) - 90000 )) + printf '%s\n' "$old_ts" > "$marker" + printf '%s\n' "$old_ts" > "$state/.wedge-permanent-$key" + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max FM_CAP_HORIZON_SECS=86400 "$WATCH" > "$out" & + pid=$! + # v11+v12: cap fires at horizon expiry, but the wedge-escalation counter was + # reset to 0 when the original cap fired. So this first poll reports + # escalation 1 (not an immediate re-fire of PERMANENTLY-WEDGED). The wedge + # re-accumulates over FM_WEDGE_MAX_ESCALATIONS polls and re-fires on the + # $max-th round. + if ! wait_for_exit "$pid" 100; then + reap "$pid"; fail "watcher did not exit on the first post-horizon poll (markers may still be short-circuiting): $(cat "$out")" + fi + grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null && fail "first post-horizon poll re-fired the cap immediately - counter was not reset when the original cap fired: $(cat "$out")" + grep -F "escalation 1" "$out" >/dev/null || fail "first post-horizon poll did not report fresh escalation count (expected escalation 1): $(cat "$out")" + ack_stopped_cycle "$state" || true + + # Drive FM_WEDGE_MAX_ESCALATIONS - 1 more polls so the wedge re-accumulates + # and re-fires the cap on the final round. + n=2 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max FM_CAP_HORIZON_SECS=86400 "$WATCH" > "$out" & + pid=$! + if ! wait_for_exit "$pid" 100; then + reap "$pid"; fail "post-horizon round $n watch failed: $(cat "$out")" + fi + if [ "$n" -lt "$max" ]; then + grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null && fail "post-horizon round $n fired PERMANENTLY-WEDGED before the cap" + else + grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null || fail "post-horizon round $max did not re-fire the cap: $(cat "$out")" + fi + ack_stopped_cycle "$state" || fail "post-horizon round $n ack failed" + n=$((n + 1)) + done + + unset FM_FAKE_CREW_STATE + pass "the cap is bound by FM_CAP_HORIZON_SECS on BOTH markers; the wedge can re-engage and re-fire on its own merits" +} + +test_wedge_cap_holds_within_horizon() { + local dir state fakebin out capture_file window key pane_hash sig pid max marker + dir=$(make_case wedge-cap-horizon-holds); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-horizon-holds" + printf 'idle wedged content' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-horizon-holds.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-horizon-holds.status" + sig=$(seen_sig "$state/wedge-cap-horizon-holds.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-horizon-holds_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle wedged content") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=3 + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + marker="$state/.wedge-permanent-$key-${pane_hash:0:12}" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_poll_cycle "$state" "$pid" || { reap "$pid"; fail "priming watch failed"; } + reap "$pid" + ack_stopped_cycle "$state" || fail "priming ack failed" + n=1 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 100 || { reap "$pid"; fail "round $n watch failed"; } + ack_stopped_cycle "$state" || fail "round $n ack failed" + n=$((n + 1)) + done + [ -e "$marker" ] || fail "cap marker missing before horizon-holds test" + + # Marker is at the cap-fire timestamp (recent, well within horizon). The cap + # MUST hold - subsequent wedge_timer_check calls must NOT re-fire the cap. + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max FM_CAP_HORIZON_SECS=86400 "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + wait "$pid" 2>/dev/null || true + ack_stopped_cycle "$state" || true + fi + reap "$pid" + ack_stopped_cycle "$state" || true + # Drain must NOT contain a stale wake for this window. + drain_out="$dir/drain.out" + FM_STATE_OVERRIDE="$state" "$DRAIN" > "$drain_out" 2>/dev/null || true + if grep "$(printf '\tstale\t')" "$drain_out" 2>/dev/null | grep -F "$window" >/dev/null; then + fail "cap re-fired within horizon (drain contains stale wake): $(cat "$drain_out")" + fi + unset FM_FAKE_CREW_STATE + pass "the cap is honored within FM_CAP_HORIZON_SECS - no additional terminal wakes fire" +} + +test_wedge_cap_window_marker_silences_fresh_hash() { + local dir state fakebin out capture_file window key pane_hash_old pane_hash_new sig pid max marker + dir=$(make_case wedge-cap-hash-change); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-hash-change" + printf 'idle wedged content' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-hash-change.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-hash-change.status" + sig=$(seen_sig "$state/wedge-cap-hash-change.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-hash-change_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash_old=$(hash_text "idle wedged content") + printf '%s' "$pane_hash_old" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=3 + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + marker="$state/.wedge-permanent-$key-${pane_hash_old:0:12}" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_poll_cycle "$state" "$pid" || { reap "$pid"; fail "priming watch failed"; } + reap "$pid" + ack_stopped_cycle "$state" || fail "priming ack failed" + n=1 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 100 || { reap "$pid"; fail "round $n watch failed"; } + ack_stopped_cycle "$state" || fail "round $n ack failed" + n=$((n + 1)) + done + [ -e "$marker" ] || fail "cap marker missing before hash-change test" + + # v12: a new hash in the SAME window is intentionally silenced by the + # window-scoped marker. v2 protected the "fresh stale hash in the same + # window" case, but that was over-eager against the hash-churning busy + # pane loop Greptile flagged on the rebased PR. Verify the new hash does + # NOT fire: the watch runs without exiting, both markers stay, and the + # drain shows no new wake. + pane_hash_new=$(hash_text "crew is alive and producing output") + printf '%s' "$pane_hash_new" > "$capture_file" + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max FM_CAP_HORIZON_SECS=86400 "$WATCH" > "$out" & + pid=$! + if wait_poll_cycle "$state" "$pid" 2>/dev/null; then + : + fi + reap "$pid" + ack_stopped_cycle "$state" || true + [ -e "$state/.wedge-permanent-$key-${pane_hash_old:0:12}" ] || fail "per-hash marker was unexpectedly cleared on hash change" + [ -e "$state/.wedge-permanent-$key" ] || fail "window-scoped marker was unexpectedly cleared on hash change" + drain_out="$dir/drain.out" + FM_STATE_OVERRIDE="$state" "$DRAIN" > "$drain_out" 2>/dev/null || true + if grep "$(printf ' stale ')" "$drain_out" 2>/dev/null | grep -F "$window" >/dev/null; then + fail "new-hash wedge was NOT suppressed by the window-scoped marker: $(cat "$drain_out")" + fi + unset FM_FAKE_CREW_STATE + pass "the window-scoped cap marker silences a fresh hash in the same window until horizon" +} + +test_wedge_cap_operator_can_rm_marker() { + local dir state fakebin out capture_file window key pane_hash sig pid max marker + dir=$(make_case wedge-cap-operator-rm); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-operator-rm" + printf 'idle wedged content' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-operator-rm.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-operator-rm.status" + sig=$(seen_sig "$state/wedge-cap-operator-rm.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-operator-rm_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle wedged content") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=3 + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + marker="$state/.wedge-permanent-$key-${pane_hash:0:12}" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_poll_cycle "$state" "$pid" || { reap "$pid"; fail "priming watch failed"; } + reap "$pid" + ack_stopped_cycle "$state" || fail "priming ack failed" + n=1 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + wait_for_exit "$pid" 100 || { reap "$pid"; fail "round $n watch failed"; } + ack_stopped_cycle "$state" || fail "round $n ack failed" + n=$((n + 1)) + done + [ -e "$marker" ] || fail "cap marker missing before operator-rm test" + + # v12: operator removes BOTH the per-hash marker AND the window-scoped + # marker to lift the cap. Removing only the per-hash leaves the window- + # scoped gate in place, which is the intended behavior (per-hash alone + # would let a hash-churning busy pane re-fire the cap on a fresh hash, + # but the window-scoped gate stops the loop). + rm -f "$marker" "$state/.wedge-permanent-$key" + + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + # v11+v12: counter is reset to 0 when the cap fires (v11), and both markers + # were rm'd above (v12). The wedge re-engages at escalation 1 and re- + # accumulates over FM_WEDGE_MAX_ESCALATIONS polls before re-firing. Verify + # the first poll reports escalation 1 (no PERMANENTLY-WEDGED), then drive + # FM_WEDGE_MAX_ESCALATIONS - 1 more polls and verify the cap re-fires on + # the $max-th round with the per-hash marker recreated. + if ! wait_for_exit "$pid" 100; then + reap "$pid"; fail "watcher did not exit on the first post-rm poll (markers may still be short-circuiting): $(cat "$out")" + fi + grep -F "escalation 1" "$out" >/dev/null || fail "first post-rm poll did not report fresh escalation count (expected escalation 1): $(cat "$out")" + grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null && fail "first post-rm poll re-fired the cap immediately - counter was not reset when the original cap fired: $(cat "$out")" + ack_stopped_cycle "$state" || true + + n=2 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_for_exit "$pid" 100; then + reap "$pid"; fail "post-rm round $n watch failed: $(cat "$out")" + fi + if [ "$n" -lt "$max" ]; then + grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null && fail "post-rm round $n fired PERMANENTLY-WEDGED before the cap" + else + grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null || fail "post-rm round $max did not re-fire the cap: $(cat "$out")" + fi + ack_stopped_cycle "$state" || fail "post-rm round $n ack failed" + n=$((n + 1)) + done + [ -e "$marker" ] || fail "per-hash marker was not recreated after operator rm re-fire" + unset FM_FAKE_CREW_STATE + pass "the operator can manually remove both cap markers for immediate re-engagement and a fresh PERMANENTLY-WEDGED wake fires once the wedge re-accumulates" +} + +test_wedge_cap_validates_invalid_override() { + local dir state fakebin out capture_file window key sig pid + dir=$(make_case wedge-cap-validate); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-validate" + printf 'idle wedged content' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-validate.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-validate.status" + sig=$(seen_sig "$state/wedge-cap-validate.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-validate_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + printf '%s' "$(hash_text "idle wedged content")" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + + # v20 (2026-09-24): FM_WEDGE_MAX_ESCALATIONS=0 is now VALID and means + # "cap disabled" (the unconfigured path). No warning logged, no cap fires + # - every escalation produces a wake. The v18 anti-pattern where 0 fell + # back to 10 with a warning is gone: 0 is the explicit opt-out, not a + # bad config. + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=0 "$WATCH" > "$out" & + pid=$! + wait_poll_cycle "$state" "$pid" || { reap "$pid"; fail "watcher with FM_WEDGE_MAX_ESCALATIONS=0 failed"; } + reap "$pid" + ack_stopped_cycle "$state" || true + # v20: 0 is valid (cap disabled), so no fallback warning. Assert no + # 'falling back' warning for the 0 case specifically. + if grep -F "FM_WEDGE_MAX_ESCALATIONS='0'" "$state/.watch-triage.log" 2>/dev/null | grep -F "falling back" >/dev/null; then + fail "FM_WEDGE_MAX_ESCALATIONS=0 should be valid (cap disabled), not produce a 'falling back' warning" + fi + grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null && fail "FM_WEDGE_MAX_ESCALATIONS=0 should NOT fire PERMANENTLY-WEDGED (cap is disabled when 0)" + + # Run with FM_WEDGE_MAX_ESCALATIONS=abc - non-integer. Default should be used. + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=abc "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + wait_for_exit "$pid" 100 || true + fi + reap "$pid" + ack_stopped_cycle "$state" || true + grep -F "FM_WEDGE_MAX_ESCALATIONS='abc'" "$state/.watch-triage.log" 2>/dev/null >/dev/null || fail "validation warning not logged for FM_WEDGE_MAX_ESCALATIONS=abc" + unset FM_FAKE_CREW_STATE + + # FM_CAP_HORIZON_SECS validation: reject 0 and non-integer. + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_CAP_HORIZON_SECS=0 "$WATCH" > "$out" & + pid=$! + wait_poll_cycle "$state" "$pid" || { reap "$pid"; fail "watcher with FM_CAP_HORIZON_SECS=0 failed"; } + reap "$pid" + ack_stopped_cycle "$state" || true + grep -F "FM_CAP_HORIZON_SECS=0" "$state/.watch-triage.log" 2>/dev/null >/dev/null || fail "validation warning not logged for FM_CAP_HORIZON_SECS=0" + + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_CAP_HORIZON_SECS=abc "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + wait_for_exit "$pid" 100 || true + fi + reap "$pid" + ack_stopped_cycle "$state" || true + grep -F "FM_CAP_HORIZON_SECS='abc'" "$state/.watch-triage.log" 2>/dev/null >/dev/null || fail "validation warning not logged for FM_CAP_HORIZON_SECS=abc" + unset FM_FAKE_CREW_STATE + pass "FM_WEDGE_MAX_ESCALATIONS=0 means cap disabled (no warning, no cap fires); FM_CAP_HORIZON_SECS rejects 0 and non-integer; FM_WEDGE_MAX_ESCALATIONS rejects non-integer (v20 semantics: opt-in by setting N>=1)" +} + +test_wedge_cap_disabled_when_zero_v20() { + # v20 (2026-09-24): the unconfigured path is now FM_WEDGE_MAX_ESCALATIONS=0 + # which means "no cap" - the pre-PR behavior, every escalation produces a + # wake. Captains opt in by setting FM_WEDGE_MAX_ESCALATIONS=N (N>=1). + # This test pins that semantics: many wedge escalations past the prior + # default of 10 do NOT write a cap marker, do NOT fire PERMANENTLY-WEDGED, + # and the escalation counter climbs past 10 unbounded. + local dir state fakebin out capture_file window key pane_hash sig pid n max marker_a ewf_after + dir=$(make_case wedge-cap-disabled-v20); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-disabled-v20" + printf 'idle wedged content' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-disabled-v20.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-disabled-v20.status" + sig=$(seen_sig "$state/wedge-cap-disabled-v20.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-disabled-v20_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle wedged content") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=15 # way past the prior default of 10 - if the cap were on, this would fire + marker_a="$state/.wedge-permanent-$key-${pane_hash:0:12}" + marker_w="$state/.wedge-permanent-$key" + + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=0 "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "priming watch failed: $(cat "$out")" + fi + reap "$pid" + ack_stopped_cycle "$state" || fail "priming ack failed" + + n=1 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=0 "$WATCH" > "$out" & + pid=$! + if ! wait_for_exit "$pid" 100; then + reap "$pid"; fail "round $n watch failed: $(cat "$out")" + fi + ack_stopped_cycle "$state" || fail "round $n ack failed" + n=$((n + 1)) + done + + # Assert NO cap markers exist (cap was off) + [ -e "$marker_a" ] && fail "per-hash cap marker written despite FM_WEDGE_MAX_ESCALATIONS=0 (cap should be disabled when 0; v20 semantics)" + [ -e "$marker_w" ] && fail "window-scoped cap marker written despite FM_WEDGE_MAX_ESCALATIONS=0 (cap should be disabled when 0; v20 semantics)" + + # Assert NO PERMANENTLY-WEDGED wake in any of the rounds + for f in $(ls -1 "$dir"/round-*-watch.out 2>/dev/null || true); do + grep -F "PERMANENTLY-WEDGED" "$f" >/dev/null && fail "PERMANENTLY-WEDGED fired in round output despite FM_WEDGE_MAX_ESCALATIONS=0 (cap should be disabled when 0; v20 semantics)" + done + # Also check the final watch output as a fallback + grep -F "PERMANENTLY-WEDGED" "$out" >/dev/null && fail "PERMANENTLY-WEDGED fired in final watch despite FM_WEDGE_MAX_ESCALATIONS=0 (cap should be disabled when 0; v20 semantics)" + + # Assert the escalation counter climbed past 10 (i.e. cap machinery ran, just didn't fire) + ewf_after=$(cat "$state/.wedge-escalations-$key" 2>/dev/null || true) + case "$ewf_after" in + ''|*[!0-9]*) fail "wedge-escalation counter missing or non-integer after $max rounds with FM_WEDGE_MAX_ESCALATIONS=0 (cap machinery not running)" ;; + *) + if [ "$ewf_after" -lt "$max" ]; then + fail "wedge-escalation counter climbed only to $ewf_after (expected $max) with FM_WEDGE_MAX_ESCALATIONS=0; cap machinery is not iterating" + fi + ;; + esac + + unset FM_FAKE_CREW_STATE + pass "FM_WEDGE_MAX_ESCALATIONS=0 disables the cap: $max escalations, no cap marker, no PERMANENTLY-WEDGED, counter climbed to $ewf_after (v20 opt-in semantics)" +} + + + +test_wedge_cap_escalation_counter_resets_on_cap_fire() { + # Regression for Greptile P1 (v11): the wedge-escalation counter in + # .wedge-escalations- MUST be reset to 0 (or removed) when the cap + # fires. Otherwise a pane that later produces a fresh hash sees n already + # at FM_WEDGE_MAX_ESCALATIONS and fires PERMANENTLY-WEDGED on the FIRST + # poll of the new hash, then on every subsequent poll. The cap exists to + # bound that exact loop; this test pins the reset by inspecting the + # counter file directly after the cap fires. + local dir state fakebin out capture_file window key pane_hash sig pid n max marker_a ewf_after_cap + dir=$(make_case wedge-cap-counter-reset); state="$dir/state"; fakebin="$dir/fakebin" + out="$dir/watch.out"; capture_file="$dir/pane.txt" + window="test:fm-wedge-cap-counter-reset" + printf 'idle wedged content' > "$capture_file" + printf 'window=%s\nkind=ship\n' "$window" > "$state/wedge-cap-counter-reset.meta" + printf 'working: still wedged\n' > "$state/wedge-cap-counter-reset.status" + sig=$(seen_sig "$state/wedge-cap-counter-reset.status"); printf '%s' "$sig" > "$state/.seen-wedge-cap-counter-reset_status" + key=$(printf '%s' "$window" | tr ':/.' '___') + pane_hash=$(hash_text "idle wedged content") + printf '%s' "$pane_hash" > "$state/.hash-$key" + printf '1\n' > "$state/.count-$key" + max=3 + export FM_FAKE_CREW_STATE='state: working · source: run-step · validating (running)' + marker_a="$state/.wedge-permanent-$key-${pane_hash:0:12}" + + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=999 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_poll_cycle "$state" "$pid"; then + reap "$pid"; fail "priming watch failed: $(cat "$out")" + fi + reap "$pid" + ack_stopped_cycle "$state" || fail "priming ack failed" + n=1 + while [ "$n" -le "$max" ]; do + echo $(( $(date +%s) - 500 )) > "$state/.stale-since-$key" + : > "$out" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_WINDOW="$window" FM_FAKE_TMUX_CAPTURE="$capture_file" \ + FM_STATE_OVERRIDE="$state" FM_CREW_STATE_BIN="$fakebin/fm-crew-state.sh" FM_STALE_ESCALATE_SECS=240 FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 FM_WEDGE_MAX_ESCALATIONS=$max "$WATCH" > "$out" & + pid=$! + if ! wait_for_exit "$pid" 100; then + reap "$pid"; fail "round $n watch failed: $(cat "$out")" + fi + ack_stopped_cycle "$state" || fail "round $n ack failed" + n=$((n + 1)) + done + [ -e "$marker_a" ] || fail "cap marker missing after firing" + + if [ -e "$state/.wedge-escalations-$key" ]; then + ewf_after_cap=$(cat "$state/.wedge-escalations-$key" 2>/dev/null || true) + case "$ewf_after_cap" in + ''|"0") ;; + *) fail "wedge-escalation counter was not reset when the cap fired (still holds '$ewf_after_cap' = saturation value) - a fresh hash will re-fire the cap immediately" + ;; + esac + fi + unset FM_FAKE_CREW_STATE + pass "the wedge-escalation counter resets to 0 when the cap fires" +} + +test_wedge_cap_fires_permanently_wedged_after_max_escalations +test_wedge_cap_suppresses_subsequent_polls_for_same_hash +test_wedge_cap_persists_across_pause_class_transitions +test_wedge_cap_expires_after_horizon +test_wedge_cap_holds_within_horizon +test_wedge_cap_window_marker_silences_fresh_hash +test_wedge_cap_escalation_counter_resets_on_cap_fire +test_wedge_cap_marker_write_after_durable_wake_v13 +test_wedge_cap_failed_window_marker_rolls_back_per_hash_v13 +test_wedge_cap_rollback_resets_state_v14 +test_wedge_cap_rollback_failure_sets_sentinel_v15 +test_wedge_cap_rollback_sentinel_keyed_on_busy_route_v17 +test_wedge_cap_window_marker_silences_hash_churning_busy_pane +test_wedge_cap_operator_can_rm_marker +test_wedge_cap_validates_invalid_override +test_wedge_cap_disabled_when_zero_v20