From b42d4fa8a752fad9a5f0235783b02534bce29219 Mon Sep 17 00:00:00 2001 From: Tiago Date: Thu, 24 Sep 2026 19:22:48 -0300 Subject: [PATCH 01/84] fix(bin): report a Lavish source armed only after its listener is running (#5566) * fix(bin): report a Lavish source armed only after its listener is running Registration alone was treated as ready, so arm could succeed before anything was collecting from the board. * no-mistakes(review): Guard Lavish arm launches, keep retire refusals, report live prior listener * no-mistakes(review): Keep polling through window before reporting a still-live prior listener * test: wait for a capture's claim to drop before the next arm The result is stored before the runner exits, so a re-arm in that gap was meeting a live claim. * no-mistakes(document): Record Lavish arm readiness evidence in verification doc * no-mistakes(ci): Both failures were caused by this PR, and both are fixed with test-only edits. Lint 2 (ShellCheck SC2034): this branch removed the only use of `reply_id` (a `start "$reply_id"` call) from tests/fm-procevent.test.sh, which left the assignment at line 1450 unused. I deleted that assignment. It was the only `reply_id` in the file. ShellCheck is now clean on both test files. Behavior portable serial 4: the failing test was tests/fm-bearings-board.test.sh, in the check "registration consumed its answer before the any-origin binding existed". I reproduced it locally: the hold was still `state: queued` when the test checked it. - What must hold: the test's check that the hold is closed must run after the listener has captured the answer. - Why it broke: the test used a stand-in adapter that ran `fm-procevent.sh start` in the foreground after `arm`, so capture finished before build returned. On this branch, `arm` starts the listener itself in the background, so the real listener captures the answer and closes the hold a moment after build returns. - Fix: removed the now-redundant stand-in adapter, the copied runtime directory, and its extra environment variables. The test now runs the real build through the existing `run_board` helper and waits up to about 10s for the hold to reach `state: done`. The checks that follow are unchanged: `Resolution mode: answered` and the any-origin binding. - Other tests: this was the only test in the file that stood in for the adapter this way. The shard's other pure-contract-unit test (tests/fm-trace-context-lib.test.sh) passed unchanged. Verification: - tests/fm-bearings-board.test.sh passed 3 times in a row via bin/fm-test-run.sh, all 18 checks, about 53s per run. - tests/fm-procevent.test.sh was not rerun, because the lint fix only removed an unused assignment --- .agents/skills/process-event-sources/SKILL.md | 4 +- bin/fm-procevent-lavish.sh | 23 +- bin/fm-procevent.sh | 82 ++++ docs/configuration.md | 6 +- docs/verification/process-event-sources.md | 1 + tests/fm-bearings-board.test.sh | 37 +- tests/fm-procevent.test.sh | 383 +++++++++++++++++- 7 files changed, 487 insertions(+), 49 deletions(-) diff --git a/.agents/skills/process-event-sources/SKILL.md b/.agents/skills/process-event-sources/SKILL.md index 8b765f01c8a..3beb9f71818 100644 --- a/.agents/skills/process-event-sources/SKILL.md +++ b/.agents/skills/process-event-sources/SKILL.md @@ -40,7 +40,9 @@ Posting that reply is best effort: a rare crash while the listener consumes the A terminal round is never re-armed: the board stays yours until you acknowledge it with `bin/fm-procevent.sh handled `, which retires it, and until then `retire` refuses the board too. Never arm a board that a live task hosts; follow the crew-hosted Lavish board contract in [`docs/configuration.md`](../../../docs/configuration.md#crew-hosted-lavish-review-boards). -Registering a source is not the same fact as listening to it: arming records the source, and a separate runner still has to pick it up. +Registering a source is not the same fact as listening to it. +Lavish `arm` waits until this registration's listener is confirmed running and does not report ready without that evidence; other adapters still record the source for the watcher's next reconcile. +When an earlier registration's listener still holds the board as the confirm window ends, Lavish `arm` prints `still-listening` instead of `armed`; that listener keeps serving the board, and the new registration takes effect only after you retire the source and arm it again. After arming by hand, confirm `bin/fm-procevent.sh list` reports that source as `live`, and run `bin/fm-procevent.sh reconcile` when it does not. Reconcile reports every launch that did not prove it took its claim within the confirm window as `failed=` and exits non-zero, so a source that cannot be started says so instead of looking armed, and it wakes you once per failure episode about it because the watcher discards that count; `start` does not fix that - if the source stays unowned, run `start` attached to read the runner's refusal, then check the source command and adapter binary the registration names, and if a later reconcile finds the source owned the episode closes on its own. A source `list` reports as `orphaned` is one reconcile will not relaunch, because something may still be polling it; reconcile wakes you once about it, and that wake's payload says which of two recoveries applies. diff --git a/bin/fm-procevent-lavish.sh b/bin/fm-procevent-lavish.sh index a99cb80aaf5..1137a477efa 100755 --- a/bin/fm-procevent-lavish.sh +++ b/bin/fm-procevent-lavish.sh @@ -198,7 +198,7 @@ cmd_source_id() { } cmd_arm() { - local artifact='' task='' reply_file='' id real + local artifact='' task='' reply_file='' id real owner listening local -a listener=() while [ "$#" -gt 0 ]; do case "$1" in @@ -239,6 +239,27 @@ cmd_arm() { FM_HOME="$FM_HOME" "$SCRIPT_DIR/fm-procevent.sh" register lavish "$id" \ -- "${listener[@]}" || exit 1 fi + # Registration is not a running listener. Readiness is the process-event + # owner's evidence for this generation; a miss retires a source that never + # started so arm does not leave it registered. + listening=0 + FM_HOME="$FM_HOME" "$SCRIPT_DIR/fm-procevent.sh" ensure-listening "$id" || listening=$? + if [ "$listening" -eq 3 ]; then + printf 'still-listening: %s\n' "$id" + printf 'artifact: %s\n' "$real" + [ -z "$task" ] || printf 'owner-task: %s\n' "$task" + printf 'note: an earlier listener is still live and serving this board; this registration takes effect only after the source is retired and armed again\n' + exit 0 + fi + if [ "$listening" -ne 0 ]; then + owner=$(FM_HOME="$FM_HOME" "$SCRIPT_DIR/fm-procevent.sh" list 2>/dev/null \ + | awk -v id="$id" '$1 == id { print $3; exit }') + case "$owner" in + live|orphaned|task:*/listening|task:*/round-open) ;; + *) FM_HOME="$FM_HOME" "$SCRIPT_DIR/fm-procevent.sh" retire "$id" >/dev/null 2>&1 || true ;; + esac + exit 1 + fi printf 'armed: %s\n' "$id" printf 'artifact: %s\n' "$real" [ -z "$task" ] || printf 'owner-task: %s\n' "$task" diff --git a/bin/fm-procevent.sh b/bin/fm-procevent.sh index a8886daf040..d8f112544d8 100755 --- a/bin/fm-procevent.sh +++ b/bin/fm-procevent.sh @@ -8,6 +8,7 @@ # fm-procevent.sh register-task -- ... # fm-procevent.sh register-extension --config-ref # fm-procevent.sh start +# fm-procevent.sh ensure-listening # fm-procevent.sh reconcile # fm-procevent.sh classify # fm-procevent.sh handled @@ -40,6 +41,15 @@ # bounded classification. Built-in results keep their existing # script command; extension results must still match the exact bound # package identity captured with them. +# ensure-listening +# Confirm the current registration generation's listener is running. +# Starts one when nothing live is in the way, and returns only after +# that generation's live claim or its launch stamp says it started. +# The wait is the reconcile confirm window and ends early on evidence. +# No evidence within the window is a nonzero result. Exit 3 means a +# live listener from another registration generation still held the +# source when the window ended, so this generation cannot start until +# it is retired. # start Claim the source, run its child to completion, durably capture the # output, publish normalized wakes for pending results, then release # the claim. It blocks for as long as the source blocks and is meant @@ -1780,6 +1790,77 @@ confirm_launched_runners() { # + local id=$1 identity=$2 state result=1 + fm_procevent_source_lock_try_acquire "$id" || return 1 + fm_procevent_claim_state_locked "$id" + state=$? + if [ "$state" -eq 0 ]; then + result=3 + [ "$FM_PROCEVENT_CLAIM_REG_IDENTITY" != "$identity" ] || result=0 + fi + fm_procevent_source_lock_release "$id" + return "$result" +} + +# 0 when no live, uncertain, leaderless, terminal, or undisplaceable claim +# blocks a launch, the same rule reconcile applies. +generation_can_launch() { # + local id=$1 state result=1 + fm_procevent_source_lock_try_acquire "$id" || return 1 + fm_procevent_claim_state_locked "$id" + state=$? + if [ "$state" -eq 1 ] && ! fm_procevent_claim_undisplaceable_locked "$id"; then + result=0 + fi + fm_procevent_source_lock_release "$id" + return "$result" +} + +# Public readiness for one source. Same evidence reconcile uses after a launch: +# a live claim bound to this registration generation, or that generation's +# launch stamp advancing. Returns as soon as either appears. A fixed sleep is +# not success. +cmd_ensure_listening() { + local id=${1-} identity before mark stamp deadline window started_once=0 listening + [ "$#" -eq 1 ] || usage + fm_procevent_source_id_valid "$id" || die "source id must be path-safe: $id" + window=$(fm_procevent_launch_confirm_seconds) \ + || die "FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS must be whole seconds from $FM_PROCEVENT_LAUNCH_CONFIRM_MIN_SECONDS to $FM_PROCEVENT_LAUNCH_CONFIRM_MAX_SECONDS" + [ -f "$(source_file "$id")" ] && [ ! -L "$(source_file "$id")" ] \ + || die "source is not registered: $id" + identity=$(fm_pr_file_identity "$(source_file "$id")" 2>/dev/null) \ + || die "cannot identify the registration: $id" + before= + if stamp=$(fm_procevent_launch_floor_stamp_path "$STATE" "$id" "$identity"); then + before=$(cat -- "$stamp" 2>/dev/null || true) + fi + deadline=$((SECONDS + 10#$window + 1)) + while :; do + listening=0 + generation_is_listening "$id" "$identity" || listening=$? + [ "$listening" -ne 0 ] || return 0 + mark= + if stamp=$(fm_procevent_launch_floor_stamp_path "$STATE" "$id" "$identity"); then + mark=$(cat -- "$stamp" 2>/dev/null || true) + fi + if [ -n "$mark" ] && [ "$mark" != "$before" ]; then + return 0 + fi + if [ "$started_once" -eq 0 ] && generation_can_launch "$id"; then + detach_runner "$id" + started_once=1 + fi + [ "$SECONDS" -lt "$deadline" ] || break + sleep 0.05 + done + [ "$listening" -ne 3 ] || return 3 + printf 'error: listener is not running: %s\n' "$id" >&2 + return 1 +} + # Stop a runner and the child it is blocked on. A runner started by reconcile is # its own process group leader, so the group signal is what actually reaches the # blocking child - signalling only the runner would leave that child alive and @@ -2339,6 +2420,7 @@ case "${1-}" in register-task) shift; cmd_register_task "$@" ;; register-extension) shift; cmd_register_extension "$@" ;; start) shift; cmd_start_public "$@" ;; + ensure-listening) shift; cmd_ensure_listening "$@" ;; _start) shift; cmd_start "$@" ;; _owner-watchdog) shift; cmd_owner_watchdog "$@" ;; reconcile) shift; cmd_reconcile "$@" ;; diff --git a/docs/configuration.md b/docs/configuration.md index 77091140c77..30f91710442 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -965,12 +965,16 @@ Before arming any Lavish source, open its artifact with `lavish-axi` so the save That adapter, and only that adapter, retries the one exact transient response a cut-short listener returns while its marks remain available (`error: Lavish Editor poll response was interrupted` with `code: SERVER_ERROR`), up to 12 times with poll starts at least 5 seconds apart, so an internal retry never reaches the runner as a captured result. This start-to-start governor is a no-op after a normally blocking poll but caps an immediately returning poll under the shipped defaults independently of the owner lease and registration launch pacing. Real feedback, ended and missing sessions, any other `SERVER_ERROR`, and that same interruption still standing once the bound is spent are all captured and announced normally; `FM_LAVISH_POLL_RETRY_DELAY` is a bounded 1 to 60 second test override for the interval only, and the runner itself stays adapter-agnostic. -An already-armed Lavish source keeps its registered listener command until it is retired and armed again, so re-arm a live board once to adopt this retry policy. +An already-armed Lavish source keeps its registered listener command until it is retired and armed again, so retire the source, then arm it again to adopt this retry policy. ### Crew-hosted Lavish review boards A live task that hosts a Lavish board owns its listener, so firstmate must never arm that board. After opening the artifact as required above, the worker arms it with `bin/fm-procevent-lavish.sh arm --for ` and never runs `lavish-axi poll` itself. +`arm` prints `armed` only after the process-event owner confirms this registration generation's listener is running, and otherwise returns nonzero without that line. +The confirmation is the same live claim or launch-stamp evidence `reconcile` already uses, bounded by `FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS`, and a failed confirmation retires a source that never started unless `retire` refuses because something may still own it, in which case the registration stays for `reconcile` or a human. +An earlier registration's listener that releases the board inside the confirm window lets the new registration start, and `arm` then reports `armed` as usual. +When a live listener from an earlier registration of the same board still holds it when the window ends, `arm` exits zero with `still-listening` instead of `armed`, because that earlier listener keeps serving the board and the new registration takes effect only after the source is retired and armed again. The arm is refused unless that task id has valid, identity-matching endpoint metadata, because a board whose owner has no endpoint would collect feedback nobody can be told about. The registration persists as one task-owned source record, while each captured nonterminal round remains open until the worker re-arms and the existing handled marker acknowledges that round. Re-arm is that acknowledgement and nothing else: the board is armed once while no record exists, and a further arm by the same owner is refused unless an unacknowledged nonterminal round is waiting, so a generation already carrying a reply is never replaced before its listener posts it. diff --git a/docs/verification/process-event-sources.md b/docs/verification/process-event-sources.md index 8abe4a71a06..3cb2af92ca7 100644 --- a/docs/verification/process-event-sources.md +++ b/docs/verification/process-event-sources.md @@ -125,6 +125,7 @@ Exercised by `tests/fm-procevent.test.sh` against a fake blocking source whose c | launch pacing during owner-loss grace | an immediately returning source that attempts detached self-relaunches is held to the configured minimum interval between command launches and remains bounded until its expired owner lease stops the generation; replacement starts a fresh pacing generation, prunes prior pacing state, and prevents a superseded sleeping runner from recreating it | | stale reclaim without displacement | concurrent contenders replacing one stale claim start exactly one runner, cross-home replacement removes the old generation's staging file from its recorded state directory, and a generation whose stale owner and independently empty process group prove it gone remains reclaimable when its recorded state-root identity can no longer be revalidated or its recorded registry directory no longer resolves to a directory, so `reconcile` reclaims it once, the replacement runs the source, and later cycles report nothing to do | | confirmed launches only | `reconcile` counts a launch as `started` only after the source is observed owned or its launch-pacing stamp has moved: a registration that cannot start is reported `failed=` with a non-zero exit and its source still listed `none`, a source that claimed, ran and exited before confirmation looked is still `started`, a zero-padded confirm window reads as base 10, and an unusable `FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS` is refused by name before any runner is launched | +| Lavish arm reports only a running listener | `bin/fm-procevent-lavish.sh arm` prints `armed` only after its own registration generation's listener has claimed the source, including when the claim is delayed by a held source lock; a registration whose runner can never claim exits non-zero after the confirm window without `armed` and is retired unless `retire` refuses; a re-arm over an earlier generation still live at the window's end exits zero with `still-listening` and starts no second listener; a re-arm whose earlier claim is released inside the window launches the new generation, which posts the worker's reply once and reports `armed`; and a stale claim with a live process group gets no second listener, a non-zero exit, and no retirement | | launch failure announced once per episode | an unconfirmed launch queues one `check` wake keyed by source, registration identity and an episode nonce; a second failure in the same episode queues nothing, a confirmed launch queues no failure and closes the episode, a later failure opens a new episode under a fresh key, and a 64-character source id keeps that key within the watcher's marker bound | | crashed leader with a live group | `SIGKILL` on only the runner leader leaves its blocking child group alive; reconcile treats that leaderless group as ambiguous, preserves its claim without starting or signalling anything, `start` runs nothing beside it, the strand is queued as one `check` wake keyed by source and claim token that a second cycle does not repeat, and reconcile still reclaims a generation with no leader and no surviving group | | reused pid with a live group | a stale claim whose recorded pid is alive under a different identity while its process group still has members is listed `orphaned`, is never relaunched by `reconcile` across cycles, is announced once naming the `start` command that clears it, and `start` reclaims it while the dead generation's leftovers can be tidied and refuses with `cannot claim source`, replacing nothing, when they cannot | diff --git a/tests/fm-bearings-board.test.sh b/tests/fm-bearings-board.test.sh index 5c37ed1a83a..6f970732a61 100644 --- a/tests/fm-bearings-board.test.sh +++ b/tests/fm-bearings-board.test.sh @@ -352,10 +352,9 @@ test_build_injects_binds_then_arms() { } test_registration_cannot_consume_before_any_origin_binding() { - local home data runtime origin key hold board sid show + local home data origin key hold board sid show home=$(make_home order-proof) data="$home/payload.json" - runtime="$home/runtime" origin=order-proof-review key=captain-choice hold="$origin-decision-$key" @@ -378,21 +377,6 @@ EOF jq --arg hold "$hold" '.captains_call[0].key = $hold' "$data" > "$data.tmp" \ && mv "$data.tmp" "$data" - mkdir -p "$runtime" - cp -R "$ROOT/bin" "$runtime/bin" - cat > "$runtime/bin/fm-procevent-lavish.sh" <<'SH' -#!/usr/bin/env bash -set -eu -if [ "${1:-}" = arm ]; then - artifact=${2:-} - "$REAL_LAVISH_ADAPTER" arm "$artifact" >/dev/null - sid=$("$REAL_LAVISH_ADAPTER" source-id "$artifact") - "$REAL_PROCEVENT" start "$sid" >/dev/null - exit 0 -fi -exec "$REAL_LAVISH_ADAPTER" "$@" -SH - chmod +x "$runtime/bin/fm-procevent-lavish.sh" cat > "$home/fakebin/lavish-axi" <<'SH' #!/usr/bin/env bash if [ -z "${1:-}" ]; then @@ -421,18 +405,17 @@ EOF SH chmod +x "$home/fakebin/lavish-axi" - PATH="$home/fakebin:$PATH" FM_ROOT_OVERRIDE="$runtime" FM_HOME="$home" \ - FM_STATE_OVERRIDE="$home/state" FM_DATA_OVERRIDE="$home/data" \ - FM_PROCEVENT_CLAIM_ROOT="$home/procevent-claims" \ - FM_BEARINGS_BOARD_TEMPLATE="$ROOT/.agents/skills/bearings/assets/board-template.html" \ - REAL_LAVISH_ADAPTER="$ROOT/bin/fm-procevent-lavish.sh" \ - REAL_PROCEVENT="$ROOT/bin/fm-procevent.sh" ORDER_PROOF_HOLD="$hold" \ - LAVISH_AXI_STATE_DIR="$home/lavish-state" \ - "$runtime/bin/fm-bearings-board.sh" build "$data" >/dev/null \ + ORDER_PROOF_HOLD="$hold" run_board "$home" build "$data" >/dev/null \ || fail "the order-proof board build failed" - show=$(cd "$home" && tasks-axi show "$hold" --full) \ - || fail "the order-proof captain hold disappeared" + # Arm starts the listener, which captures the answer and closes the hold on + # its own schedule after build returns. + for _ in $(seq 1 100); do + show=$(cd "$home" && tasks-axi show "$hold" --full) \ + || fail "the order-proof captain hold disappeared" + case "$show" in *"state: done"*) break ;; esac + sleep 0.1 + done assert_contains "$show" "state: done" \ "registration consumed its answer before the any-origin binding existed" assert_contains "$show" "Resolution mode: answered" \ diff --git a/tests/fm-procevent.test.sh b/tests/fm-procevent.test.sh index b653512e7bb..9d91f9a2ba8 100755 --- a/tests/fm-procevent.test.sh +++ b/tests/fm-procevent.test.sh @@ -156,6 +156,23 @@ wait_for() { # [tries] return 1 } +# Arm now starts the listener, so a later start would poll again. Wait for the +# capture that listener is already producing, and for its runner to release the +# claim: the result lands before the runner publishes and exits, and a retire or +# re-arm in that gap meets a live claim the synchronous start never left behind. +wait_capture() { # [tries] + local home=$1 id=$2 n=${3:-100} + local _ + for _ in $(seq 1 "$n"); do + if first_result "$home" "$id" >/dev/null 2>&1 \ + && [ ! -e "$FM_PROCEVENT_CLAIM_ROOT/$id.claim" ]; then + return 0 + fi + sleep 0.1 + done + return 1 +} + # [tries]: wait until holds at least lines. A # detached runner appends its execution marker after the command that started it # has already returned, so a caller that needs that append must wait for it @@ -163,7 +180,11 @@ wait_for() { # [tries] wait_for_lines() { local f=$1 want=$2 n=${3:-100} have for _ in $(seq 1 "$n"); do - have=$(wc -l < "$f" 2>/dev/null | tr -d ' ') + if [ -f "$f" ]; then + have=$(wc -l < "$f" | tr -d ' ') + else + have=0 + fi case "$have" in ''|*[!0-9]*) have=0 ;; esac [ "$have" -ge "$want" ] && return 0 sleep 0.1 @@ -803,8 +824,8 @@ fi assert_contains "$(cat "$MULTI_ROOT/firstmate-arm.err")" "owned by task worker-1" \ "second armer refusal did not name the worker owner" list_out=$(FM_HOME="$HMULTI" "$ROOT/bin/fm-procevent.sh" list) -assert_contains "$list_out" "task:worker-1/dead" \ - "the source list did not expose the worker-owned board state" +assert_contains "$list_out" "task:worker-1/listening" \ + "arm did not leave the worker-owned board with a live listener" PATH="$MULTI_BIN:$PATH" LAVISH_AXI_HOST=recovery.example LAVISH_AXI_PORT=34387 FM_HOME="$HMULTI" \ pe "$HMULTI" start "$multi_id" > "$MULTI_ROOT/run1" 2>&1 & MULTI_RUN=$! @@ -948,8 +969,11 @@ pass "worker-owned Lavish rounds deliver to the worker, acknowledge on re-arm, a # the same sequence and still routes to the owning worker. HORPHAN="$TMP_ROOT/horphan"; new_home "$HORPHAN" ORPHAN_BIN=$(fm_fakebin "$TMP_ROOT/lavish-orphan-stub") +ORPHAN_TRIGGER="$TMP_ROOT/lavish-orphan-hold" +export ORPHAN_TRIGGER cat > "$ORPHAN_BIN/lavish-axi" <<'SH' #!/usr/bin/env bash +while [ ! -e "$ORPHAN_TRIGGER" ]; do sleep 0.02; done printf 'session:\n status: feedback\nprompts[1]{uid,prompt,selector,tag,text}:\n "","after the crash","","message",""\n' SH chmod +x "$ORPHAN_BIN/lavish-axi" @@ -965,7 +989,12 @@ PATH="$ORPHAN_BIN:$PATH" FM_HOME="$HORPHAN" \ chmod 0700 "$HORPHAN/state/procevent-inbox" printf 'worker-4\n' > "$HORPHAN/state/procevent-inbox/$orphan_id.1.owner-task" chmod 0600 "$HORPHAN/state/procevent-inbox/$orphan_id.1.owner-task" +touch "$ORPHAN_TRIGGER" PATH="$ORPHAN_BIN:$PATH" pe "$HORPHAN" start "$orphan_id" >/dev/null 2>&1 || true +wait_for "$HORPHAN/state/procevent-inbox/$orphan_id.1.result" \ + || fail "an owner sidecar with no committed result wedged the next capture of its source" +wait_for "$HORPHAN/state/worker-4.inbox/001.msg" \ + || fail "the recovered capture did not reach its owning worker's steering inbox" [ -f "$HORPHAN/state/procevent-inbox/$orphan_id.1.result" ] \ || fail "an owner sidecar with no committed result wedged the next capture of its source" [ -f "$HORPHAN/state/worker-4.inbox/001.msg" ] \ @@ -992,7 +1021,8 @@ fm_test_track_procevent_home "$HADOPT" new_task_endpoint "$HADOPT" worker-5 PATH="$ADOPT_BIN:$PATH" FM_HOME="$HADOPT" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$ADOPT_ART" >/dev/null -PATH="$ADOPT_BIN:$PATH" pe "$HADOPT" start "$adopt_id" >/dev/null 2>&1 || true +wait_capture "$HADOPT" "$adopt_id" \ + || fail "the firstmate fixture capture never landed" [ -f "$HADOPT/state/procevent-inbox/$adopt_id.1.result" ] \ || fail "the firstmate fixture capture never landed" [ ! -f "$HADOPT/state/procevent-inbox/$adopt_id.1.handled" ] \ @@ -1052,7 +1082,8 @@ fm_test_track_procevent_home "$HREDELIVER" new_task_endpoint "$HREDELIVER" worker-6 PATH="$ADOPT_BIN:$PATH" FM_HOME="$HREDELIVER" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$REDELIVER_ART" --for worker-6 >/dev/null -PATH="$ADOPT_BIN:$PATH" pe "$HREDELIVER" start "$redeliver_id" >/dev/null 2>&1 || true +wait_capture "$HREDELIVER" "$redeliver_id" \ + || fail "the first worker-owned round was never captured" [ -f "$HREDELIVER/state/worker-6.inbox/001.msg" ] \ || fail "the first worker-owned round never reached the worker inbox" mv "$HREDELIVER/state/worker-6.inbox/001.msg" \ @@ -1083,7 +1114,8 @@ fm_test_track_procevent_home "$HCONC" new_task_endpoint "$HCONC" worker-7 PATH="$CONC_BIN:$PATH" FM_HOME="$HCONC" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$CONC_ART" --for worker-7 >/dev/null -PATH="$CONC_BIN:$PATH" pe "$HCONC" start "$conc_id" >/dev/null 2>&1 || true +wait_capture "$HCONC" "$conc_id" \ + || fail "the terminal worker-owned round never landed" [ -f "$HCONC/state/procevent-inbox/$conc_id.1.result" ] \ || fail "the terminal worker-owned round never landed" [ -e "$HCONC/state/procevent/$conc_id.source" ] \ @@ -1137,7 +1169,8 @@ fm_test_track_procevent_home "$HINTR" new_task_endpoint "$HINTR" worker-12 PATH="$INTR_BIN:$PATH" FM_HOME="$HINTR" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$INTR_ART" --for worker-12 >/dev/null -PATH="$INTR_BIN:$PATH" pe "$HINTR" start "$intr_id" >/dev/null 2>&1 || true +wait_capture "$HINTR" "$intr_id" \ + || fail "the terminal worker-owned round was never captured" [ "$(cat "$INTR_ROOT/count" 2>/dev/null || echo 0)" = 1 ] \ || fail "the terminal worker-owned round was not polled exactly once" rm -f "$HINTR/state/procevent/$intr_id.source" @@ -1183,9 +1216,15 @@ printf 'reply from generation two\n' > "$ROLL_ROOT/reply2" PATH="$ROLL_BIN:$PATH" FM_HOME="$HROLL" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$ROLL_ART" --for worker-8 \ --agent-reply-file "$ROLL_ROOT/reply1" >/dev/null -PATH="$ROLL_BIN:$PATH" pe "$HROLL" start "$roll_id" >/dev/null 2>&1 || true +wait_for "$ROLL_ROOT/replies" \ + || fail "the first generation's reply never reached the board" [ "$(grep -c 'generation one' "$ROLL_ROOT/replies" 2>/dev/null || true)" = 1 ] \ || fail "the first generation's reply never reached the board" +# The reply is posted before the round is captured. Making the inbox read-only +# before the runner commits and exits would fail that capture instead of the +# re-arm's acknowledgement, leaving no round for the retried re-arm. +wait_capture "$HROLL" "$roll_id" \ + || fail "the first generation's round was never captured" cp "$HROLL/state/procevent/$roll_id.source" "$ROLL_ROOT/generation-one.source" chmod 0500 "$HROLL/state/procevent-inbox" rollback_status=0 @@ -1202,7 +1241,8 @@ cmp -s "$ROLL_ROOT/generation-one.source" "$HROLL/state/procevent/$roll_id.sourc PATH="$ROLL_BIN:$PATH" FM_HOME="$HROLL" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$ROLL_ART" --for worker-8 \ --agent-reply-file "$ROLL_ROOT/reply2" >/dev/null -PATH="$ROLL_BIN:$PATH" pe "$HROLL" start "$roll_id" >/dev/null 2>&1 || true +wait_for_lines "$ROLL_ROOT/replies" 2 \ + || fail "the retried re-arm did not hand the board its generation's reply exactly once" [ "$(grep -c 'generation two' "$ROLL_ROOT/replies" 2>/dev/null || true)" = 1 ] \ || fail "the retried re-arm did not hand the board its generation's reply exactly once" pass "a re-arm that cannot acknowledge its round leaves the running generation alone" @@ -1219,6 +1259,7 @@ cat > "$REARM_BIN/lavish-axi" <<'SH' #!/usr/bin/env bash set -eu [ "${3-}" != --agent-reply ] || printf '%s\n' "$4" >> "$REARM_ROOT/replies" +while [ ! -e "$REARM_ROOT/release" ]; do sleep 0.02; done printf 'session:\n status: feedback\nprompts[1]{uid,prompt,selector,tag,text}:\n "","one more round","","message",""\n' SH chmod +x "$REARM_BIN/lavish-axi" @@ -1242,7 +1283,11 @@ if PATH="$REARM_BIN:$PATH" FM_HOME="$HREARM" \ fi assert_contains "$(cat "$REARM_ROOT/idle-rearm.err")" "worker-11" \ "the refused idle re-arm did not name the task that already holds the board" +touch "$REARM_ROOT/release" PATH="$REARM_BIN:$PATH" pe "$HREARM" start "$rearm_id" >/dev/null 2>&1 || true +wait_for "$REARM_ROOT/replies" || fail "the listener never posted the reply it was armed with" +wait_for "$HREARM/state/procevent-inbox/$rearm_id.1.result" \ + || fail "the first worker-owned round never landed" [ "$(grep -c 'first generation reply' "$REARM_ROOT/replies" 2>/dev/null || true)" = 1 ] \ || fail "the refused idle re-arm cost the board the reply its listener was already carrying" [ -f "$HREARM/state/procevent-inbox/$rearm_id.1.result" ] \ @@ -1259,7 +1304,8 @@ PATH="$REARM_BIN:$PATH" FM_HOME="$HREARM" \ --agent-reply-file "$REARM_ROOT/reply2" >/dev/null [ -f "$HREARM/state/procevent-inbox/$rearm_id.1.handled" ] \ || fail "re-arming over an open round did not acknowledge that round" -PATH="$REARM_BIN:$PATH" pe "$HREARM" start "$rearm_id" >/dev/null 2>&1 || true +wait_for_lines "$REARM_ROOT/replies" 2 \ + || fail "the acknowledging re-arm did not hand the board its own generation's reply" [ "$(grep -c 'second generation reply' "$REARM_ROOT/replies" 2>/dev/null || true)" = 1 ] \ || fail "the acknowledging re-arm did not hand the board its own generation's reply" pass "a worker-owned board is armed once and re-armed only to acknowledge an open round" @@ -1401,7 +1447,6 @@ HREPLY="$TMP_ROOT/hreply"; new_home "$HREPLY" REPLY_ART="$TMP_ROOT/reply-retry-board.html" printf '

reply retry

\n' > "$REPLY_ART" lavish_session "$REPLY_ART" -reply_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$REPLY_ART") fm_test_track_procevent_home "$HREPLY" new_task_endpoint "$HREPLY" worker-9 printf 'applied round one\n' > "$TMP_ROOT/reply-retry.txt" @@ -1410,7 +1455,8 @@ LAVISH_COUNT="$TMP_ROOT/reply-retry-count"; LAVISH_SCRIPT="interrupt interrupt f PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HREPLY" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$REPLY_ART" --for worker-9 \ --agent-reply-file "$TMP_ROOT/reply-retry.txt" >/dev/null -PATH="$LAVISH_SCRIPTED_BIN:$PATH" pe "$HREPLY" start "$reply_id" >/dev/null +wait_for "$HREPLY/state/worker-9.inbox/001.msg" 200 \ + || fail "the round that delivered after quiet retries did not reach the worker inbox" [ "$(cat "$LAVISH_COUNT")" = 3 ] \ || fail "the reply-carrying listener was polled $(cat "$LAVISH_COUNT") times, not the two quiet retries plus the delivering poll" [ "$(grep -c 'applied round one' "$LAVISH_REPLY_LOG" 2>/dev/null || true)" = 1 ] \ @@ -1492,11 +1538,14 @@ fm_test_track_procevent_home "$HEXH" LAVISH_COUNT="$TMP_ROOT/exhaust-count"; LAVISH_SCRIPT="interrupt" PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HEXH" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$EXH_ART" >/dev/null -PATH="$LAVISH_SCRIPTED_BIN:$PATH" pe "$HEXH" start "$exh_id" >/dev/null +wait_capture "$HEXH" "$exh_id" 200 \ + || fail "exhaustion produced no captured result" [ "$(cat "$LAVISH_COUNT")" = 13 ] \ || fail "the retry bound polled $(cat "$LAVISH_COUNT") times, not the first poll plus 12 bounded retries" [ "$(count_results "$HEXH" "$exh_id")" = 1 ] \ || fail "exhaustion produced $(count_results "$HEXH" "$exh_id") captured results instead of one" +wait_for "$HEXH/state/.wake-queue" \ + || fail "the interruption that survives the bound produced no wake" assert_contains "$(wake_payloads "$HEXH")" "procevent lavish $exh_id 1" \ "the interruption that survives the bound is announced normally" assert_grep 'poll response was interrupted' "$(first_result "$HEXH" "$exh_id")" \ @@ -1516,7 +1565,8 @@ fm_test_track_procevent_home "$HOTHER" LAVISH_COUNT="$TMP_ROOT/other-count"; LAVISH_SCRIPT="other-server-error" PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HOTHER" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$OTHER_ART" >/dev/null -PATH="$LAVISH_SCRIPTED_BIN:$PATH" pe "$HOTHER" start "$other_id" >/dev/null +wait_for "$HOTHER/state/.wake-queue" \ + || fail "an unrelated SERVER_ERROR is captured and announced immediately" [ "$(cat "$LAVISH_COUNT")" = 1 ] \ || fail "an unrelated SERVER_ERROR was retried $(cat "$LAVISH_COUNT") times instead of surfacing at once" assert_contains "$(wake_payloads "$HOTHER")" "procevent lavish $other_id 1" \ @@ -1537,7 +1587,8 @@ fm_test_track_procevent_home "$HNEAR" LAVISH_COUNT="$TMP_ROOT/near-count"; LAVISH_SCRIPT="near-interrupt feedback" PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HNEAR" FM_LAVISH_POLL_RETRY_DELAY=1 \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$NEAR_ART" >/dev/null -PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HNEAR" pe "$HNEAR" start "$near_id" >/dev/null +wait_for "$HNEAR/state/.wake-queue" \ + || fail "a whitespace variant of the interruption is captured and announced immediately" [ "$(cat "$LAVISH_COUNT")" = 1 ] \ || fail "a near-match interruption was retried instead of surfacing on its first poll" assert_contains "$(wake_payloads "$HNEAR")" "procevent lavish $near_id 1" \ @@ -1589,11 +1640,10 @@ lavish_session "$STREAM_ART" stream_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$STREAM_ART") fm_test_track_procevent_home "$HSTREAM" LAVISH_COUNT="$TMP_ROOT/stream-count"; LAVISH_SCRIPT="stream" -PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HSTREAM" \ - "$ROOT/bin/fm-procevent-lavish.sh" arm "$STREAM_ART" >/dev/null -PATH="$LAVISH_SCRIPTED_BIN:$PATH" TMPDIR="$STREAM_TMPDIR" \ +PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HSTREAM" TMPDIR="$STREAM_TMPDIR" \ LAVISH_STREAM_READY="$LAVISH_STREAM_READY" LAVISH_STREAM_RELEASE="$LAVISH_STREAM_RELEASE" \ - FM_PROCEVENT_MAX_OUTPUT_BYTES=100 pe "$HSTREAM" reconcile >/dev/null + FM_PROCEVENT_MAX_OUTPUT_BYTES=100 \ + "$ROOT/bin/fm-procevent-lavish.sh" arm "$STREAM_ART" >/dev/null wait_for "$LAVISH_STREAM_READY" || fail "streaming poll did not start" stream_staged=("$STREAM_TMPDIR"/fm-lavish-poll.*) [ -e "${stream_staged[0]}" ] || fail "streaming poll created no classifier staging file" @@ -4346,4 +4396,299 @@ kill -0 -"$CRASH_PID" 2>/dev/null \ pass "a group whose leader died to something else is still refused, not signalled" kill -KILL -"$CRASH_PID" 2>/dev/null || true +# --- arm reports ready only once this registration's listener is running ---- +# The public arm path used to print armed as soon as registration was stored. +# A listener that has not claimed the source is not ready, so arm waits for the +# same live-claim or launch-stamp evidence reconcile uses and fails closed when +# that evidence does not appear within the confirm window. +READY="$TMP_ROOT/ready-arm" +mkdir -p "$READY/bin" "$READY/home/state" +cat > "$READY/bin/lavish-axi" <<'SH' +#!/usr/bin/env bash +printf 'started\n' >> "${READY_MARK:?}" +while [ ! -e "${READY_RELEASE:?}" ]; do sleep 0.02; done +printf 'session:\n status: ended\n' +SH +chmod +x "$READY/bin/lavish-axi" +ready_art="$READY/board.html" +printf '

ready

\n' > "$ready_art" +lavish_session "$ready_art" +ready_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$ready_art") +fm_test_track_procevent_home "$READY/home" +export READY_MARK="$READY/mark" READY_RELEASE="$READY/release" +: > "$READY_MARK" +PATH="$READY/bin:$PATH" FM_HOME="$READY/home" \ + "$ROOT/bin/fm-procevent-lavish.sh" arm "$ready_art" > "$READY/arm.out" +assert_contains "$(cat "$READY/arm.out")" "armed: $ready_id" "a live listener was not reported ready" +[ -e "$FM_PROCEVENT_CLAIM_ROOT/$ready_id.claim" ] \ + || fail "arm reported ready without a listener claim" +for _ in $(seq 1 50); do + grep -q started "$READY_MARK" && break + sleep 0.05 +done +grep -q started "$READY_MARK" || fail "arm reported ready before the listener command ran" +touch "$READY_RELEASE" +for _ in $(seq 1 50); do + [ -e "$FM_PROCEVENT_CLAIM_ROOT/$ready_id.claim" ] || break + sleep 0.05 +done +PATH="$READY/bin:$PATH" FM_HOME="$READY/home" \ + "$ROOT/bin/fm-procevent-lavish.sh" retire "$ready_art" >/dev/null 2>&1 || true +pass "arm reports ready only after the listener is running" + +# Delayed start: the source lock is held so the listener cannot claim, and arm +# must not print armed until that lock clears and the listener does. +DELAY="$TMP_ROOT/delay-arm" +mkdir -p "$DELAY/bin" "$DELAY/home/state" +cp "$READY/bin/lavish-axi" "$DELAY/bin/lavish-axi" +delay_art="$DELAY/board.html" +printf '

delay

\n' > "$delay_art" +lavish_session "$delay_art" +delay_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$delay_art") +fm_test_track_procevent_home "$DELAY/home" +export READY_MARK="$DELAY/mark" READY_RELEASE="$DELAY/release" +: > "$READY_MARK" +delay_ready="$DELAY/lock-ready" +delay_rel="$DELAY/lock-release" +hold_source_lock "$delay_id" "$delay_ready" "$delay_rel" +wait_for "$delay_ready" || fail "delayed-start fixture could not hold the source lock" +PATH="$DELAY/bin:$PATH" FM_HOME="$DELAY/home" \ + "$ROOT/bin/fm-procevent-lavish.sh" arm "$delay_art" > "$DELAY/arm.out" 2>"$DELAY/arm.err" & +delay_arm=$! +sleep 0.4 +assert_not_contains "$(cat "$DELAY/arm.out" 2>/dev/null || true)" "armed:" \ + "arm reported ready while the listener could not start" +[ ! -e "$FM_PROCEVENT_CLAIM_ROOT/$delay_id.claim" ] \ + || fail "a listener claimed the source while its lock was held" +touch "$delay_rel" +wait "$delay_arm" || fail "arm failed after the delayed listener was allowed to start: $(cat "$DELAY/arm.err")" +assert_contains "$(cat "$DELAY/arm.out")" "armed: $delay_id" \ + "arm did not report ready once the delayed listener was running" +for _ in $(seq 1 50); do + grep -q started "$READY_MARK" && break + sleep 0.05 +done +grep -q started "$READY_MARK" || fail "the delayed listener never ran" +touch "$READY_RELEASE" +wait "$HOLDER_PID" 2>/dev/null || true +for _ in $(seq 1 50); do + [ -e "$FM_PROCEVENT_CLAIM_ROOT/$delay_id.claim" ] || break + sleep 0.05 +done +PATH="$DELAY/bin:$PATH" FM_HOME="$DELAY/home" \ + "$ROOT/bin/fm-procevent-lavish.sh" retire "$delay_art" >/dev/null 2>&1 || true +pass "arm waits out a delayed listener start before reporting ready" + +# A claim path that is a directory can never be owned, so the runner dies before +# the listener command. Arm must not print ready, and it must remove the +# registration it just published. +arm_blocked_claim() { # + local dir=$1 secs=$2 art id began rc elapsed + mkdir -p "$dir/bin" "$dir/home/state" + cat > "$dir/bin/lavish-axi" <<'SH' +#!/bin/sh +printf started >> "${READY_MARK:?}" +SH + chmod +x "$dir/bin/lavish-axi" + art="$dir/board.html" + printf '

blocked

\n' > "$art" + lavish_session "$art" + id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$art") + fm_test_track_procevent_home "$dir/home" + mkdir -p "$FM_PROCEVENT_CLAIM_ROOT/$id.claim" + export READY_MARK="$dir/mark" + : > "$READY_MARK" + began=$(date +%s) + set +e + PATH="$dir/bin:$PATH" FM_HOME="$dir/home" FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS="$secs" \ + "$ROOT/bin/fm-procevent-lavish.sh" arm "$art" > "$dir/arm.out" 2>"$dir/arm.err" + rc=$? + set -e + elapsed=$(( $(date +%s) - began )) + [ "$rc" -ne 0 ] || fail "arm reported success when no listener could claim ($dir)" + assert_not_contains "$(cat "$dir/arm.out")" "armed:" \ + "arm printed ready when no listener could claim ($dir)" + [ ! -s "$READY_MARK" ] || fail "the listener command ran without a claim ($dir)" + # retire refuses a claim it cannot read, and arm must not override it. + [ -e "$dir/home/state/procevent/$id.source" ] \ + || fail "arm removed a registration that retire refused to remove ($dir)" + printf '%s\n' "$elapsed" > "$dir/elapsed" + rmdir "$FM_PROCEVENT_CLAIM_ROOT/$id.claim" 2>/dev/null || true + PATH="$dir/bin:$PATH" FM_HOME="$dir/home" \ + "$ROOT/bin/fm-procevent-lavish.sh" retire "$art" >/dev/null 2>&1 || true +} + +arm_blocked_claim "$TMP_ROOT/immediate-arm" 1 +pass "arm fails when the listener cannot claim, and leaves the registration retire refused" + +arm_blocked_claim "$TMP_ROOT/timeout-arm" 2 +tout_elapsed=$(cat "$TMP_ROOT/timeout-arm/elapsed") +[ "$tout_elapsed" -ge 2 ] \ + || fail "arm did not wait out the confirm window (${tout_elapsed}s)" +pass "arm waits out the confirm window before reporting that the listener is not running" + +# Re-arming a firstmate-owned board publishes a new registration while the +# earlier generation's listener still holds the claim. When that listener still +# holds it as the confirm window ends, it keeps serving the board, so arm must +# say so instead of reporting failure, and must never claim this generation is +# the one listening. +LIVE="$TMP_ROOT/live-rearm" +mkdir -p "$LIVE/bin" "$LIVE/home/state" +cp "$READY/bin/lavish-axi" "$LIVE/bin/lavish-axi" +live_art="$LIVE/board.html" +printf '

live

\n' > "$live_art" +lavish_session "$live_art" +live_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$live_art") +fm_test_track_procevent_home "$LIVE/home" +export READY_MARK="$LIVE/mark" READY_RELEASE="$LIVE/release" +: > "$READY_MARK" +PATH="$LIVE/bin:$PATH" FM_HOME="$LIVE/home" \ + "$ROOT/bin/fm-procevent-lavish.sh" arm "$live_art" > "$LIVE/arm1.out" +assert_contains "$(cat "$LIVE/arm1.out")" "armed: $live_id" "the first arm was not reported ready" +wait_for_lines "$READY_MARK" 1 || fail "the first generation's listener never ran" +set +e +PATH="$LIVE/bin:$PATH" FM_HOME="$LIVE/home" FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS=1 \ + "$ROOT/bin/fm-procevent-lavish.sh" arm "$live_art" > "$LIVE/arm2.out" 2> "$LIVE/arm2.err" +live_rc=$? +set -e +[ "$live_rc" -eq 0 ] \ + || fail "re-arm over a live earlier listener failed ($live_rc): $(cat "$LIVE/arm2.err")" +assert_contains "$(cat "$LIVE/arm2.out")" "still-listening: $live_id" \ + "re-arm did not say the earlier listener is still serving the board" +assert_contains "$(cat "$LIVE/arm2.out")" "retired and armed again" \ + "re-arm did not say how the new registration takes effect" +assert_not_contains "$(cat "$LIVE/arm2.out")" "armed: $live_id" \ + "re-arm reported ready for a registration whose own listener is not running" +assert_not_contains "$(cat "$LIVE/arm2.err")" "error:" \ + "re-arm over a live earlier listener printed an error" +[ "$(pe "$LIVE/home" list | awk -v id="$live_id" '$1 == id { print $3 }')" = live ] \ + || fail "re-arm disturbed the live earlier listener" +sleep 0.3 +[ "$(wc -l < "$READY_MARK" | tr -d ' ')" = 1 ] \ + || fail "re-arm started a second listener beside the live earlier one" +touch "$READY_RELEASE" +PATH="$LIVE/bin:$PATH" FM_HOME="$LIVE/home" \ + "$ROOT/bin/fm-procevent-lavish.sh" retire "$live_art" >/dev/null 2>&1 || true +pass "re-arm over a live earlier listener reports it still serving the board" + +# A worker re-arms as soon as its round is published, which can land while the +# earlier generation's runner is still finishing and holding the claim. Once +# that claim is released inside the confirm window, arm must start the new +# generation carrying the worker's reply and report it armed. +DRAIN="$TMP_ROOT/draining-rearm" +mkdir -p "$DRAIN/bin" "$DRAIN/home/state" +export DRAIN +cat > "$DRAIN/bin/lavish-axi" <<'SH' +#!/usr/bin/env bash +set -eu +[ "${3-}" != --agent-reply ] || printf '%s\n' "$4" >> "$DRAIN/replies" +printf 'poll\n' >> "$DRAIN/polls" +if [ "$(wc -l < "$DRAIN/polls")" -ge 2 ]; then + while [ ! -e "$DRAIN/release2" ]; do sleep 0.02; done +else + while [ ! -e "$DRAIN/release1" ]; do sleep 0.02; done +fi +printf 'session:\n status: feedback\nprompts[1]{uid,prompt,selector,tag,text}:\n "","next round","","message",""\n' +SH +chmod +x "$DRAIN/bin/lavish-axi" +drain_art="$DRAIN/board.html" +printf '

drain

\n' > "$drain_art" +lavish_session "$drain_art" +drain_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$drain_art") +fm_test_track_procevent_home "$DRAIN/home" +new_task_endpoint "$DRAIN/home" worker-drain +printf 'first drain reply\n' > "$DRAIN/reply1" +printf 'second drain reply\n' > "$DRAIN/reply2" +PATH="$DRAIN/bin:$PATH" FM_HOME="$DRAIN/home" \ + "$ROOT/bin/fm-procevent-lavish.sh" arm "$drain_art" --for worker-drain \ + --agent-reply-file "$DRAIN/reply1" >/dev/null \ + || fail "the first generation of the draining fixture did not arm" +drain_claim="$FM_PROCEVENT_CLAIM_ROOT/$drain_id.claim" +cp "$drain_claim" "$DRAIN/generation-one.claim" +touch "$DRAIN/release1" +wait_for "$DRAIN/home/state/procevent-inbox/$drain_id.1.result" \ + || fail "the first generation of the draining fixture never captured its round" +for _ in $(seq 1 100); do + [ -e "$drain_claim" ] || break + sleep 0.05 +done +[ ! -e "$drain_claim" ] || fail "the first generation of the draining fixture never exited" +# Stand the first generation's claim back up on a live process so the re-arm +# meets it still held, then release it partway through the confirm window. +setsid sleep 60 & +drain_holder=$! +drain_holder_identity=$(bash -c '. "$1/bin/fm-wake-lib.sh"; fm_pid_identity "$2"' _ "$ROOT" "$drain_holder") \ + || fail "could not read the draining holder's identity" +awk -v pid="$drain_holder" -v ident="$drain_holder_identity" \ + 'NR == 2 { print pid; next } NR == 4 { print ident; next } { print }' \ + "$DRAIN/generation-one.claim" > "$drain_claim" +chmod 0600 "$drain_claim" +[ "$(pe "$DRAIN/home" list | awk -v id="$drain_id" '$1 == id { print $3 }')" = task:worker-drain/round-open ] \ + || fail "fixture invalid: the stood-up first generation is not reported live: $(pe "$DRAIN/home" list)" +PATH="$DRAIN/bin:$PATH" FM_HOME="$DRAIN/home" FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS=5 \ + "$ROOT/bin/fm-procevent-lavish.sh" arm "$drain_art" --for worker-drain \ + --agent-reply-file "$DRAIN/reply2" > "$DRAIN/arm2.out" 2> "$DRAIN/arm2.err" & +drain_arm=$! +sleep 1 +kill -KILL "$drain_holder" 2>/dev/null || true +wait "$drain_holder" 2>/dev/null || true +wait "$drain_arm" \ + || fail "re-arm failed after the earlier claim was released: $(cat "$DRAIN/arm2.err")" +assert_contains "$(cat "$DRAIN/arm2.out")" "armed: $drain_id" \ + "re-arm did not launch the new generation once the earlier claim was released" +assert_not_contains "$(cat "$DRAIN/arm2.out")" "still-listening" \ + "re-arm reported the released earlier listener as still serving the board" +wait_for_lines "$DRAIN/replies" 2 \ + || fail "the new generation never handed the board the worker's reply" +[ "$(grep -c 'second drain reply' "$DRAIN/replies" 2>/dev/null || true)" = 1 ] \ + || fail "the new generation did not hand the board its own reply exactly once" +touch "$DRAIN/release2" +wait_for "$DRAIN/home/state/procevent-inbox/$drain_id.2.result" \ + || fail "the new generation never captured its round" +pass "re-arm launches the new generation once a draining earlier claim is released" + +# A stale claim whose process group is still alive may still have its polling +# child on the board's session. Reconcile refuses to launch beside it, and arm +# must apply the same rule instead of adding a second destructive poller. +UNDISP="$TMP_ROOT/undisplaceable-arm" +mkdir -p "$UNDISP/bin" "$UNDISP/home/state" +cp "$READY/bin/lavish-axi" "$UNDISP/bin/lavish-axi" +undisp_art="$UNDISP/board.html" +printf '

undisplaceable

\n' > "$undisp_art" +lavish_session "$undisp_art" +undisp_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$undisp_art") +fm_test_track_procevent_home "$UNDISP/home" +export READY_MARK="$UNDISP/mark" READY_RELEASE="$UNDISP/release" +: > "$READY_MARK" +PATH="$UNDISP/bin:$PATH" FM_HOME="$UNDISP/home" \ + "$ROOT/bin/fm-procevent-lavish.sh" arm "$undisp_art" >/dev/null +wait_for_lines "$READY_MARK" 1 || fail "the undisplaceable fixture's listener never ran" +undisp_claim="$FM_PROCEVENT_CLAIM_ROOT/$undisp_id.claim" +undisp_identity=$(sed -n '4p' "$undisp_claim") +awk 'NR == 4 { print "different-live-process-identity"; next } { print }' \ + "$undisp_claim" > "$undisp_claim.tmp" && mv "$undisp_claim.tmp" "$undisp_claim" +chmod 0600 "$undisp_claim" +[ "$(pe "$UNDISP/home" list | awk -v id="$undisp_id" '$1 == id { print $3 }')" = orphaned ] \ + || fail "fixture invalid: the reused-pid claim is not reported orphaned" +set +e +PATH="$UNDISP/bin:$PATH" FM_HOME="$UNDISP/home" FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS=1 \ + "$ROOT/bin/fm-procevent-lavish.sh" arm "$undisp_art" > "$UNDISP/arm2.out" 2>/dev/null +undisp_rc=$? +set -e +sleep 0.5 +[ "$(wc -l < "$READY_MARK" | tr -d ' ')" = 1 ] \ + || fail "arm started a second listener beside a stale claim's live process group" +[ "$undisp_rc" -ne 0 ] || fail "arm reported success beside an undisplaceable claim" +assert_not_contains "$(cat "$UNDISP/arm2.out")" "armed: $undisp_id" \ + "arm reported ready beside an undisplaceable claim" +[ -e "$UNDISP/home/state/procevent/$undisp_id.source" ] \ + || fail "arm retired a source whose earlier listener may still be polling" +awk -v v="$undisp_identity" 'NR == 4 { print v; next } { print }' \ + "$undisp_claim" > "$undisp_claim.tmp" && mv "$undisp_claim.tmp" "$undisp_claim" +chmod 0600 "$undisp_claim" +touch "$READY_RELEASE" +PATH="$UNDISP/bin:$PATH" FM_HOME="$UNDISP/home" \ + "$ROOT/bin/fm-procevent-lavish.sh" retire "$undisp_art" >/dev/null 2>&1 || true +pass "arm does not launch beside a stale claim whose process group is alive" + printf '\nall procevent tests passed\n' From 52e679b6f6889872aee18f1be07c35f02e55dfd8 Mon Sep 17 00:00:00 2001 From: Tiago Date: Thu, 24 Sep 2026 23:20:25 -0300 Subject: [PATCH 02/84] fix(bin): stop repeating unknown-wake escalations that were already delivered (#5599) * fix(bin): acknowledge a delivered unknown-wake escalation The same unrecognized wake was escalated again after it had already been handled, because delivery never recorded that identity. * no-mistakes(review): Scope unknown-wake acknowledgements to one away session * no-mistakes(review): Clear delivered digest when unknown-wake ack write fails * no-mistakes(review): Limit unknown-wake suppression to acknowledged lines * no-mistakes(document): List unknown-wake ack file among away-session artifacts --- .agents/skills/afk/SKILL.md | 8 +++- bin/fm-afk-launch.sh | 9 ++-- bin/fm-afk-return.sh | 3 +- bin/fm-afk-start.sh | 3 +- bin/fm-supervise-daemon.sh | 44 +++++++++++++++++-- tests/fm-afk-launch.test.sh | 12 +++-- tests/fm-afk-return.test.sh | 2 + tests/fm-daemon.test.sh | 88 +++++++++++++++++++++++++++++++++++++ 8 files changed, 154 insertions(+), 15 deletions(-) diff --git a/.agents/skills/afk/SKILL.md b/.agents/skills/afk/SKILL.md index 65b402f14c3..a15790c5e3f 100644 --- a/.agents/skills/afk/SKILL.md +++ b/.agents/skills/afk/SKILL.md @@ -178,7 +178,11 @@ Classify each wake this way: Healthy crewmates are autonomous and do not wait on firstmate mid-task. - `heartbeat` -> self-handle. The daemon runs its own cheap bash fleet scan every `FM_HEARTBEAT_SCAN_SECS` (default 300s) as the catch-all for captain-relevant events still unread by the per-wake classifier. -- An unknown wake reason escalates fail-safe, while status-read uncertainty follows the shared one-report-without-position-advance contract referenced under Dedupe below. +- An unknown wake reason escalates fail-safe. + After that escalation is delivered, its exact distilled line is acknowledged and the same identity does not escalate again during that away session. + A new away session starts with no acknowledgements, so a handled identity can present once more. + An identity that was not delivered still escalates. + Status-read uncertainty follows the shared one-report-without-position-advance contract referenced under Dedupe below. Escalations are buffered up to `FM_ESCALATE_BATCH_SECS` (default 90s; 0 = immediate) and flushed as one single-line digest prefixed with the current @@ -239,7 +243,7 @@ the operational prefix lets firstmate distinguish it from a real captain message ### Stale-artifact lifecycle -Treat `state/.subsuper-escalations`, its `.since` sidecar, and `state/.subsuper-inject-wedged` as session-scoped delivery artifacts, not as the durable work record. +Treat `state/.subsuper-escalations`, its `.since` sidecar, `state/.subsuper-inject-wedged`, and `state/.subsuper-unknown-acked` as session-scoped delivery artifacts, not as the durable work record. Always enter through `bin/fm-afk-launch.sh`, which clears prior-session artifacts only for a fresh entry and preserves the current session's buffer on refresh. Always exit through `bin/fm-afk-launch.sh stop`, which keeps `state/.afk` present through the daemon's shutdown flush, clears it, and archives the posture record last. `docs/herdr-backend.md` "Away-mode supervisor support" owns the current mechanism, and `docs/verification/runtime-backends.md` "Away-mode transport" owns active evidence. diff --git a/bin/fm-afk-launch.sh b/bin/fm-afk-launch.sh index 829d362146c..93a54feaef3 100755 --- a/bin/fm-afk-launch.sh +++ b/bin/fm-afk-launch.sh @@ -487,11 +487,12 @@ fm_afk_launch_restore_backup() { # rm -f "$FM_AFK_LAUNCH_STATE/.afk" \ "$FM_AFK_LAUNCH_STATE/.subsuper-escalations" \ "$FM_AFK_LAUNCH_STATE/.subsuper-escalations.since" \ - "$FM_AFK_LAUNCH_STATE/.subsuper-inject-wedged" || result=1 + "$FM_AFK_LAUNCH_STATE/.subsuper-inject-wedged" \ + "$FM_AFK_LAUNCH_STATE/.subsuper-unknown-acked" || result=1 if [ "$had_afk" -eq 1 ]; then cp "$backup/.afk" "$FM_AFK_LAUNCH_STATE/.afk" || result=1 fi - for artifact in .subsuper-escalations .subsuper-escalations.since .subsuper-inject-wedged; do + for artifact in .subsuper-escalations .subsuper-escalations.since .subsuper-inject-wedged .subsuper-unknown-acked; do if [ -e "$backup/$artifact" ]; then cp -p "$backup/$artifact" "$FM_AFK_LAUNCH_STATE/$artifact" || result=1 fi @@ -615,7 +616,7 @@ fm_afk_launch_start() { had_afk=1 cp "$FM_AFK_LAUNCH_STATE/.afk" "$backup/.afk" || { rm -rf "$backup"; return 1; } fi - for artifact in .subsuper-escalations .subsuper-escalations.since .subsuper-inject-wedged; do + for artifact in .subsuper-escalations .subsuper-escalations.since .subsuper-inject-wedged .subsuper-unknown-acked; do if [ -e "$FM_AFK_LAUNCH_STATE/$artifact" ]; then cp -p "$FM_AFK_LAUNCH_STATE/$artifact" "$backup/$artifact" || { rm -rf "$backup"; return 1; } fi @@ -672,7 +673,7 @@ fm_afk_launch_start_native() { had_afk=1 cp "$FM_AFK_LAUNCH_STATE/.afk" "$backup/.afk" || { rm -rf "$backup"; return 1; } fi - for artifact in .subsuper-escalations .subsuper-escalations.since .subsuper-inject-wedged; do + for artifact in .subsuper-escalations .subsuper-escalations.since .subsuper-inject-wedged .subsuper-unknown-acked; do if [ -e "$FM_AFK_LAUNCH_STATE/$artifact" ]; then cp -p "$FM_AFK_LAUNCH_STATE/$artifact" "$backup/$artifact" || { rm -rf "$backup"; return 1; } fi diff --git a/bin/fm-afk-return.sh b/bin/fm-afk-return.sh index 93dcd1f8684..7ae898847a4 100755 --- a/bin/fm-afk-return.sh +++ b/bin/fm-afk-return.sh @@ -257,7 +257,8 @@ clear_delivery_artifacts() { rm -f \ "$STATE/.subsuper-escalations" \ "$STATE/.subsuper-escalations.since" \ - "$STATE/.subsuper-inject-wedged" + "$STATE/.subsuper-inject-wedged" \ + "$STATE/.subsuper-unknown-acked" } # The lifecycle retention reasons the gate kept, one per line, empty when the diff --git a/bin/fm-afk-start.sh b/bin/fm-afk-start.sh index e268d2d61e0..3ddafc5107f 100755 --- a/bin/fm-afk-start.sh +++ b/bin/fm-afk-start.sh @@ -63,7 +63,8 @@ fm_afk_clear_stale_artifacts() { # local state=$1 rm -f "$state/.subsuper-escalations" \ "$state/.subsuper-escalations.since" \ - "$state/.subsuper-inject-wedged" 2>/dev/null + "$state/.subsuper-inject-wedged" \ + "$state/.subsuper-unknown-acked" 2>/dev/null } daemon_lock_owner() { diff --git a/bin/fm-supervise-daemon.sh b/bin/fm-supervise-daemon.sh index 472d19a20cb..ba959cf2f18 100755 --- a/bin/fm-supervise-daemon.sh +++ b/bin/fm-supervise-daemon.sh @@ -466,11 +466,40 @@ classify_heartbeat() { printf 'self|heartbeat (catch-all scan runs in housekeeping)' } -# Anything unrecognized is escalated (fail-safe). +# Anything unrecognized is escalated (fail-safe). A delivered unknown wake is +# acknowledged by its exact distilled line in state/.subsuper-unknown-acked, so +# that same identity does not escalate again in this away session; the away +# entry and return paths clear that file. An identity still only buffered, +# or never successfully flushed, is not acknowledged and still escalates. classify_unknown() { # printf 'escalate|unknown wake: %s' "$1" } +# Exact distilled line of an unknown-wake escalation, or nothing. +unknown_wake_line() { # + case "$1" in + "unknown wake: "*) printf '%s' "$1"; return 0 ;; + esac + return 1 +} + +unknown_wake_acknowledged() { # + local ack="$1/.subsuper-unknown-acked" + [ -f "$ack" ] || return 1 + grep -Fxq -- "$2" "$ack" +} + +# Record every unknown-wake line from a flush that already reached the supervisor. +# Ordinary escalation lines are left alone. +unknown_wake_acknowledge_flushed() { # + local state=$1 buf=$2 line + while IFS= read -r line || [ -n "$line" ]; do + unknown_wake_line "$line" >/dev/null || continue + unknown_wake_acknowledged "$state" "$line" && continue + printf '%s\n' "$line" >> "$state/.subsuper-unknown-acked" || return 1 + done < "$buf" +} + # --- stale marker + escalation buffer (stateful, but via explicit state dir) - # Marker: state/.subsuper-stale- contains the epoch first seen idle. # Buffer: state/.subsuper-escalations one distilled line per escalation. @@ -693,7 +722,10 @@ stale_window_is_busy() { # } escalate_add() { # - local state=$1 item=$2 buf + local state=$1 item=$2 buf line + if line=$(unknown_wake_line "$item"); then + unknown_wake_acknowledged "$state" "$line" && return 0 + fi buf="$state/.subsuper-escalations" [ -s "$buf" ] || _now > "${buf}.since" printf '%s\n' "$item" >> "$buf" @@ -712,7 +744,13 @@ escalate_flush() { # # Single-line wrapper: no embedded newlines (inject_msg also collapses as a # safety net, but keeping the source single-line makes the intent explicit). msg=$(printf 'Supervisor escalate (%s event(s)): %s (pre-read; re-arm not needed — watcher daemon-managed)' "$n" "$msg") - if inject_msg "$msg" "$state"; then : > "$buf"; rm -f "${buf}.since" "$state/.subsuper-inject-wedged"; return 0; fi + if inject_msg "$msg" "$state"; then + unknown_wake_acknowledge_flushed "$state" "$buf" \ + || log "unknown-wake acknowledgement write failed; a delivered unknown wake may escalate again" + : > "$buf" + rm -f "${buf}.since" "$state/.subsuper-inject-wedged" + return 0 + fi return 1 } diff --git a/tests/fm-afk-launch.test.sh b/tests/fm-afk-launch.test.sh index 7a8e435e775..6a0356f19d1 100755 --- a/tests/fm-afk-launch.test.sh +++ b/tests/fm-afk-launch.test.sh @@ -218,7 +218,7 @@ unit_stop_archives_the_record_last() { } # --------------------------------------------------------------------------- -# UNIT 1: fm_afk_clear_stale_artifacts removes exactly the three stale artifacts. +# UNIT 1: fm_afk_clear_stale_artifacts removes exactly the four stale artifacts. # --------------------------------------------------------------------------- unit_clear_stale() { local st @@ -227,6 +227,7 @@ unit_clear_stale() { : > "$st/state/.subsuper-escalations" : > "$st/state/.subsuper-escalations.since" : > "$st/state/.subsuper-inject-wedged" + : > "$st/state/.subsuper-unknown-acked" : > "$st/state/.wake-queue" # durable queue must be untouched # Source fm-afk-start.sh inside a child bash (it sets `set -eu` and would # otherwise leak that into this test shell) and call the clear helper. @@ -234,8 +235,9 @@ unit_clear_stale() { bash -c '. "$1"; fm_afk_clear_stale_artifacts "$2"' _ "$START" "$st/state" if [ ! -e "$st/state/.subsuper-escalations" ] \ && [ ! -e "$st/state/.subsuper-escalations.since" ] \ - && [ ! -e "$st/state/.subsuper-inject-wedged" ]; then - pass "clear-stale: removes escalations buffer, sidecar, and wedge marker" + && [ ! -e "$st/state/.subsuper-inject-wedged" ] \ + && [ ! -e "$st/state/.subsuper-unknown-acked" ]; then + pass "clear-stale: removes escalations buffer, sidecar, wedge marker, and unknown-wake acknowledgements" else fail "clear-stale: stale artifacts survived" fi @@ -305,6 +307,7 @@ unit_fresh_vs_refresh() { mkdir -p "$st/state" : > "$st/state/.subsuper-escalations" : > "$st/state/.subsuper-inject-wedged" + : > "$st/state/.subsuper-unknown-acked" # A live "daemon": a real process whose identity the lock records, so # daemon_lock_held_by_live_daemon returns true (a refresh). sleep 600 & @@ -315,7 +318,8 @@ unit_fresh_vs_refresh() { # shellcheck source=/dev/null ( . "$ROOT/bin/fm-wake-lib.sh"; fm_pid_identity "$sleep_pid" > "$lock/pid-identity" 2>/dev/null ) || true FM_HOME="$st" FM_STATE_OVERRIDE="$st/state" "$START" >/dev/null 2>&1 - if [ -e "$st/state/.subsuper-escalations" ] && [ -e "$st/state/.subsuper-inject-wedged" ]; then + if [ -e "$st/state/.subsuper-escalations" ] && [ -e "$st/state/.subsuper-inject-wedged" ] \ + && [ -e "$st/state/.subsuper-unknown-acked" ]; then pass "refresh: daemon already alive - stale artifacts preserved (current session's buffer kept)" else fail "refresh: incorrectly cleared the current session's buffered escalations" diff --git a/tests/fm-afk-return.test.sh b/tests/fm-afk-return.test.sh index 0ce90aba151..89c729caedc 100755 --- a/tests/fm-afk-return.test.sh +++ b/tests/fm-afk-return.test.sh @@ -116,6 +116,7 @@ test_return_gate_owns_remediation_and_reports_catchup_to_bearings() { date +%s > "$dir/home/state/.afk" printf 'repair-task.status: blocked synthetic dependency\n' > "$dir/home/state/.subsuper-escalations" printf 'fm away-mode inject WEDGED: 4555s undelivered\n' > "$dir/home/state/.subsuper-inject-wedged" + printf 'unknown wake: frobnicate: handled\n' > "$dir/home/state/.subsuper-unknown-acked" { printf '1784074271\t2\tsignal\trepair-task.status\tsignal: synthetic status\n' printf 'wake annotation: latest wake-EVENT observed at drain, not current state: repair-task.status: blocked synthetic dependency\n' @@ -195,6 +196,7 @@ test_return_gate_owns_remediation_and_reports_catchup_to_bearings() { [ ! -e "$gate" ] || fail "successful check left the return gate behind" [ ! -e "$dir/home/state/.subsuper-escalations" ] || fail "successful check left delivered escalation state behind" [ ! -e "$dir/home/state/.subsuper-inject-wedged" ] || fail "successful check left the wedge marker behind" + [ ! -e "$dir/home/state/.subsuper-unknown-acked" ] || fail "successful check left the away session's unknown-wake acknowledgements behind" [ -s "$dir/home/state/.fake-drain" ] || fail "successful return consumed its wake before handling completed" [ ! -e "$dir/home/state/.fake-drain-acks" ] || fail "successful return acknowledged its wake inside evidence publication" assert_contains "$out" 'WAKE_ACK_REQUIRED: after handling completes' "successful return did not hand acknowledgement to the handling turn" diff --git a/tests/fm-daemon.test.sh b/tests/fm-daemon.test.sh index 510a4320988..74ec0905a10 100755 --- a/tests/fm-daemon.test.sh +++ b/tests/fm-daemon.test.sh @@ -605,6 +605,92 @@ test_classify_check_and_unknown_escalate() { pass "check + unknown escalate; heartbeat self-handles" } +# An unrecognized wake escalates once per identity. Delivery acknowledges that +# exact line; a later copy does not escalate again. A different identity still +# escalates, and an identity that never flushed still escalates. Ordinary +# escalation lines are not part of that acknowledgement. A new away session +# clears the acknowledgements, so the same identity can fire again. +test_unknown_wake_ack_suppresses_handled_identity() { + local dir state fakebin sent capture out + dir=$(make_supercase unknown-wake-ack) + state="$dir/state" + fakebin="$dir/fakebin" + sent="$dir/sent.log"; : > "$sent" + capture="$dir/pane.txt"; printf '\342\235\257 \n' > "$capture" + + FM_ESCALATE_BATCH_SECS=999 handle_wake "frobnicate: already-handled" "$state" \ + || fail "the first unknown wake was not handled" + [ ! -e "$state/.subsuper-unknown-acked" ] \ + || fail "an undelivered unknown wake was acknowledged" + + : > "$state/.subsuper-escalations" + FM_ESCALATE_BATCH_SECS=999 handle_wake "frobnicate: already-handled" "$state" \ + || fail "an undelivered unknown wake did not escalate again after its buffer was lost" + [ "$(grep -c 'unknown wake: frobnicate: already-handled' "$state/.subsuper-escalations")" = 1 ] \ + || fail "a lost undelivered unknown wake did not escalate again" + + afk_enter "$state" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_PANE_ALIVE=1 FM_FAKE_TMUX_SENT="$sent" \ + FM_FAKE_TMUX_CAPTURE="$capture" FM_ESCALATE_BATCH_SECS=0 escalate_flush "$state" \ + || fail "unknown-wake flush failed" + grep -F 'unknown wake: frobnicate: already-handled' "$state/.subsuper-unknown-acked" >/dev/null \ + || fail "a delivered unknown wake was not acknowledged" + [ ! -s "$state/.subsuper-escalations" ] || fail "delivered unknown wake stayed buffered" + + FM_ESCALATE_BATCH_SECS=999 handle_wake "frobnicate: already-handled" "$state" \ + || fail "an acknowledged unknown wake was not handled" + [ ! -s "$state/.subsuper-escalations" ] \ + || fail "an acknowledged unknown wake escalated again: $(cat "$state/.subsuper-escalations")" + + FM_ESCALATE_BATCH_SECS=999 handle_wake "frobnicate: brand-new" "$state" \ + || fail "a new unknown wake was not handled" + out=$(cat "$state/.subsuper-escalations" 2>/dev/null || true) + case "$out" in + "unknown wake: frobnicate: brand-new") ;; + *) fail "a new unknown wake did not escalate on its own: $out" ;; + esac + escalate_add "$state" "done: PR https://example.test/pull/9" + [ "$(grep -c 'done: PR https://example.test/pull/9' "$state/.subsuper-escalations")" = 1 ] \ + || fail "an ordinary escalation was swallowed by unknown-wake acknowledgement" + escalate_add "$state" "done: PR https://example.test/pull/9" + [ "$(grep -c 'done: PR https://example.test/pull/9' "$state/.subsuper-escalations")" = 2 ] \ + || fail "an ordinary escalation was deduped by unknown-wake acknowledgement" + + bash -c '. "$1"; fm_afk_clear_stale_artifacts "$2"' _ "$AFK_START" "$state" \ + || fail "clearing the away-session artifacts failed" + FM_ESCALATE_BATCH_SECS=999 handle_wake "frobnicate: already-handled" "$state" \ + || fail "an unknown wake from a prior session was not handled" + [ "$(grep -c 'unknown wake: frobnicate: already-handled' "$state/.subsuper-escalations")" = 1 ] \ + || fail "an unknown wake acknowledged in a prior away session did not fire again" + pass "a delivered unknown wake is acknowledged once per away session; a new one and ordinary escalations still fire" +} + +# A digest that inject_msg already delivered must not be injected again just +# because the acknowledgement write failed afterwards. +test_unknown_wake_ack_failure_still_clears_delivered_digest() { + local dir state fakebin sent capture + dir=$(make_supercase unknown-wake-ack-failure) + state="$dir/state" + fakebin="$dir/fakebin" + sent="$dir/sent.log"; : > "$sent" + capture="$dir/pane.txt"; printf '\342\235\257 \n' > "$capture" + mkdir -p "$state/.subsuper-unknown-acked" + + FM_ESCALATE_BATCH_SECS=999 handle_wake "frobnicate: ack-write-fails" "$state" \ + || fail "the unknown wake was not handled" + escalate_add "$state" "done: PR https://example.test/pull/10" + afk_enter "$state" + PATH="$fakebin:$PATH" FM_FAKE_TMUX_PANE_ALIVE=1 FM_FAKE_TMUX_SENT="$sent" \ + FM_FAKE_TMUX_CAPTURE="$capture" FM_ESCALATE_BATCH_SECS=0 escalate_flush "$state" 2>/dev/null \ + || fail "a delivered digest was reported undelivered after its acknowledgement write failed" + grep -F 'unknown wake: frobnicate: ack-write-fails' "$sent" >/dev/null \ + || fail "the digest was not delivered: $(cat "$sent")" + [ ! -s "$state/.subsuper-escalations" ] \ + || fail "a delivered digest stayed buffered for re-injection: $(cat "$state/.subsuper-escalations")" + [ ! -e "$state/.subsuper-escalations.since" ] || fail "a delivered digest kept its batch timer" + pass "a failed unknown-wake acknowledgement write does not re-inject a delivered digest" +} + test_stale_transient_self_records_marker() { local dir state out key dir=$(make_supercase stale-transient) @@ -2788,6 +2874,8 @@ test_daemon_state_root_uses_fm_home test_classify_routine_signal_self test_classify_terminal_signal_escalates test_classify_check_and_unknown_escalate +test_unknown_wake_ack_suppresses_handled_identity +test_unknown_wake_ack_failure_still_clears_delivered_digest test_stale_transient_self_records_marker test_stale_diagnostic_wedge_survives_busy_housekeeping test_enriched_wedge_under_declared_wait_uses_pause_cadence From c6e816ffadcc62e7d98f60b8483eda8dc3acc527 Mon Sep 17 00:00:00 2001 From: Tiago Date: Thu, 24 Sep 2026 23:20:56 -0300 Subject: [PATCH 03/84] fix(bin): keep a stated default-key retraction from cancelling a keyless wait (#5587) * fix(bin): keep a stated default retraction from cancelling a keyless wait A resolved line that names the shared default decision bucket was closing the keyless live wait that only prints as that same key. Keyless self-retraction still closes the keyless wait. * no-mistakes(review): Keep declared waits standing past foreign-key resolved lines * no-mistakes(review): Bound declared-wait read and share one decision-key parser * no-mistakes(document): Document supervisors' key-aware declared-wait read --- bin/fm-classify-lib.sh | 124 +++++++++++++++++++++---- bin/fm-push-transition-lib.sh | 2 +- bin/fm-supervise-daemon.sh | 23 ++--- bin/fm-watch.sh | 20 ++-- docs/architecture.md | 3 +- tests/fm-classify-decision-key.test.sh | 77 +++++++++++++++ tests/fm-daemon.test.sh | 24 +++++ tests/fm-watch-triage.test.sh | 35 +++++++ 8 files changed, 268 insertions(+), 40 deletions(-) diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index 509c80f8012..f33c4039722 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -163,24 +163,30 @@ last_status_line() { # [] # A bare legacy free-text line counts as an event only when a captain token leads # it, so continuation prose that merely mentions one cannot hide a declaration. _fm_status_event_scan() { - local line last='' prev='' fallback='' verb legacy_re unstamped + local line last='' prev='' fallback='' legacy_re legacy_re="^[[:space:]]*(${FM_CAPTAIN_RE:-$FM_CLASSIFY_CAPTAIN_RE_DEFAULT})" while IFS= read -r line || [ -n "$line" ]; do case "$line" in *[![:space:]]*) fallback=$line ;; *) continue ;; esac - case "$line" in *:*) status_line_verb "$line" verb ;; *) verb='' ;; esac - case "$verb" in - working|needs-decision|blocked|done|failed|note|\ - "${FM_CLASSIFY_PAUSED_VERB:-$FM_CLASSIFY_PAUSED_VERB_DEFAULT}"|\ - "${FM_CLASSIFY_RESOLVE_VERB:-$FM_CLASSIFY_RESOLVE_VERB_DEFAULT}"|\ - "${FM_CLASSIFY_CAPTAIN_HELD_VERB:-$FM_CLASSIFY_CAPTAIN_HELD_VERB_DEFAULT}") prev=$last; last=$line ;; - *) _fm_status_unstamped "$line" unstamped - _fm_classify_matches "$unstamped" "$legacy_re" && { prev=$last; last=$line; } ;; - esac + _fm_status_line_is_event "$line" "$legacy_re" && { prev=$last; last=$line; } done printf '%s\n%s\n' "$prev" "${last:-$fallback}" [ -n "$last" ] } +# 0 when a nonblank is a recognized status event for the scan above. +_fm_status_line_is_event() { # + local verb unstamped + case "$1" in *:*) status_line_verb "$1" verb ;; *) verb='' ;; esac + case "$verb" in + working|needs-decision|blocked|done|failed|note|\ + "${FM_CLASSIFY_PAUSED_VERB:-$FM_CLASSIFY_PAUSED_VERB_DEFAULT}"|\ + "${FM_CLASSIFY_RESOLVE_VERB:-$FM_CLASSIFY_RESOLVE_VERB_DEFAULT}"|\ + "${FM_CLASSIFY_CAPTAIN_HELD_VERB:-$FM_CLASSIFY_CAPTAIN_HELD_VERB_DEFAULT}") return 0 ;; + esac + _fm_status_unstamped "$1" unstamped + _fm_classify_matches "$unstamped" "$2" +} + # 0 when matches the extended regex case-insensitively, leaving # the caller's nocasematch setting untouched. _fm_classify_matches() { # @@ -267,6 +273,66 @@ status_is_paused_or_captain_held() { # status_is_paused "$line" || status_is_captain_held "$line" } +# The status line that holds a crew in a declared wait, or nothing when it is in +# none. Supervisors decide the wait from this line, never from the raw latest +# event: a resolved line is also how firstmate answers a decision (fm-send +# --resolve-key), and one that lands after a pause for a different phase key - +# including the stated default key a keyless decision shares - does not end the +# pause. Only a resolved line for the pause's own phase key (the keyed +# activity fold's key, where a keyless line is its own phase) retracts it, as +# does any other later event. A captain-held line counts only while it is the +# latest event. Bounded like last_status_line: only a tail window made wholly of +# resolved events widens the read to the whole file. +status_declared_wait_line() { # + local f=$1 last verb resolve legacy_re + last=$(last_status_line "$f") + if status_is_paused_or_captain_held "$last"; then + printf '%s\n' "$last" + return 0 + fi + resolve=${FM_CLASSIFY_RESOLVE_VERB:-$FM_CLASSIFY_RESOLVE_VERB_DEFAULT} + status_line_verb "$last" verb + [ "$verb" = "$resolve" ] || return 0 + legacy_re="^[[:space:]]*(${FM_CAPTAIN_RE:-$FM_CLASSIFY_CAPTAIN_RE_DEFAULT})" + tail -n "$FM_CLASSIFY_EVENT_WINDOW_LINES" "$f" 2>/dev/null \ + | _fm_status_declared_wait_scan "$resolve" "$legacy_re" \ + || _fm_status_declared_wait_scan "$resolve" "$legacy_re" < "$f" || : +} + +# Walk the status lines on stdin back from the newest event past resolved lines +# to the first other event, and print it when it is a pause none of those +# resolved lines share a phase key with. Returns 1 when every event is a +# resolved line, so a caller reading a bounded window knows to widen it. +_fm_status_declared_wait_scan() { # + local resolve=$1 legacy_re=$2 line verb key keys=$'\n' i=0 + local -a lines=() + while IFS= read -r line || [ -n "$line" ]; do + lines[i]=$line + i=$((i + 1)) + done + while [ "$i" -gt 0 ]; do + i=$((i - 1)) + line=${lines[i]} + case "$line" in *[![:space:]]*) ;; *) continue ;; esac + _fm_status_line_is_event "$line" "$legacy_re" || continue + status_line_verb "$line" verb + case "$verb" in + "$resolve") ;; + "${FM_CLASSIFY_PAUSED_VERB:-$FM_CLASSIFY_PAUSED_VERB_DEFAULT}") ;; + *) return 0 ;; + esac + key=$(_fm_decision_key "$line" "$_FM_CLASSIFY_KEYLESS_PHASE") || key= + if [ "$verb" = "$resolve" ]; then + keys="$keys$key"$'\n' + continue + fi + case "$keys" in *$'\n'"$key"$'\n'*) return 0 ;; esac + printf '%s\n' "$line" + return 0 + done + return 1 +} + # A condition-aware declared wait: a `paused:` line may say WHEN it expects to # clear with `until ` anywhere in its text (UTC only, so # no local-zone guess is ever recorded). Prints that time as epoch seconds so a @@ -587,7 +653,7 @@ status_line_note() { # -> text after the first colon, trimmed fi printf '%s' "$n" } -_fm_decision_key() { # -> key slug, or "default" when no token +_fm_decision_key() { # [] -> key slug, or (default "default") when no token local k unstamped _fm_status_unstamped "$1" unstamped if _fm_key_before_colon "$unstamped"; then @@ -595,7 +661,7 @@ _fm_decision_key() { # -> key slug, or "default" when no token k=${k#*\[key=} k=${k%%\]*} else - k=$(_fm_key_at_note_head "$unstamped") || { printf 'default'; return 0; } + k=$(_fm_key_at_note_head "$unstamped") || { printf '%s' "${2-default}"; return 0; } fi _fm_decision_slug_ok "$k" || return 1 printf '%s' "$k" @@ -758,8 +824,8 @@ status_open_decisions() { # [] # Resolve the log's current declaration at one boundary for crew-state consumers. # Any decision the fold still holds open wins over unrelated events, and the -# fold's most recently opened record supplies it; the latest recognized event -# stands when nothing is open. +# fold's most recently opened record supplies it; a standing declared wait, then +# the latest recognized event, stands when nothing is open. # Actual run/pane evidence is still reconciled by fm-crew-state.sh. status_current_line() { # local open key verb note current='' @@ -769,6 +835,7 @@ status_current_line() { # done < + local line key rest + while IFS= read -r line; do + [ -n "$line" ] || continue + key=${line%%$'\t'*} + rest=${line#*$'\t'} + [ "$key" = "$_FM_CLASSIFY_KEYLESS_PHASE" ] && key=default + printf '%s\t%s\n' "$key" "$rest" + done < diff --git a/bin/fm-push-transition-lib.sh b/bin/fm-push-transition-lib.sh index 497cdc0d68b..12f87d78abb 100644 --- a/bin/fm-push-transition-lib.sh +++ b/bin/fm-push-transition-lib.sh @@ -150,7 +150,7 @@ handle_push_transition() { # # external dependency, or the captain a verified hold transferred the work to. # Either way the wait is durably recorded, so absorb the immediate escalation # and leave the bounded re-surface to the watcher's own pause cadence. - if status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")"; then + if status_is_paused_or_captain_held "$(status_declared_wait_line "$STATE/$task.status")"; then triage_log "absorbed push $to (declared wait, awaiting external or captain): $window" fm_backend_commit_transition "$backend" "$STATE" "$session" "$record" || exit 1 return diff --git a/bin/fm-supervise-daemon.sh b/bin/fm-supervise-daemon.sh index ba959cf2f18..f047e99e4a4 100755 --- a/bin/fm-supervise-daemon.sh +++ b/bin/fm-supervise-daemon.sh @@ -406,7 +406,7 @@ classify_signal() { # # first sight of a non-terminal stale it returns "self" and the caller records a # timestamp marker; persistence is escalated by housekeeping's recheck, not here. classify_stale() { # [ ] - local win=$1 state=$2 record=${3-} rc=${4-} task last event rest + local win=$1 state=$2 record=${3-} rc=${4-} task last declared event rest task=$(window_to_task "$win" "$state") if [ -z "$rc" ]; then record=$(status_span_first_actionable_record "$state/$task.status" \ @@ -424,14 +424,15 @@ classify_stale() { # [ ] printf 'escalate|stale + actionable status: %s' "$event" return fi - if [ -n "$last" ] && status_is_paused_or_captain_held "$last"; then + declared=$(status_declared_wait_line "$state/$task.status") + if [ -n "$declared" ] && status_is_paused_or_captain_held "$declared"; then # A DECLARED external-wait pause or a verified captain-held transfer # (fm-classify-lib.sh owns which declarations qualify): an idle pane is # EXPECTED, so this is not a wedge. The caller records a pause marker (long # re-surface cadence in housekeeping) rather than a wedge stale marker. Cheap: - # reuses the status line already read, no fm-crew-state.sh call, mirroring the + # a status-file read, no fm-crew-state.sh call, mirroring the # daemon's existing status-log classification. - printf 'pause|paused (awaiting external), rechecked on a long cadence: %s' "$last" + printf 'pause|paused (awaiting external), rechecked on a long cadence: %s' "$declared" return fi if [ -n "$last" ] && status_is_captain_relevant "$last"; then @@ -577,7 +578,7 @@ migrate_watcher_pause_markers() { # task=$(basename "$meta"); task=${task%.meta} key=$(_stale_key "$task") watcher_key=$(_stale_key "$win") - last=$(last_status_line "$state/$task.status") + last=$(status_declared_wait_line "$state/$task.status") if status_is_paused_or_captain_held "$last" || [ -e "$state/.subsuper-paused-$key" ] || [ -e "$state/.paused-$watcher_key" ]; then reconcile_pause_tracking "$win" "$state" "$last" fi @@ -591,7 +592,7 @@ sync_pause_markers_from_signal() { # for f in "${files[@]}"; do case "$f" in *.status) ;; *) continue ;; esac [ -e "$f" ] || continue - last=$(last_status_line "$f") + last=$(status_declared_wait_line "$f") task=$(basename "$f"); task=${task%.status} win=$(window_for_task "$task" "$state" 2>/dev/null || true) [ -n "$win" ] || continue @@ -1104,7 +1105,7 @@ housekeeping() { # rm -f "$marker"; continue fi task=$(window_to_task "$win" "$state") - last=$(last_status_line "$state/$task.status") + last=$(status_declared_wait_line "$state/$task.status") if [ -n "$last" ] && status_is_paused_or_captain_held "$last"; then reconcile_pause_tracking "$win" "$state" "$last" continue @@ -1145,7 +1146,7 @@ housekeeping() { # rm -f "$marker"; continue fi task=$(window_to_task "$win" "$state") - last=$(last_status_line "$state/$task.status") + last=$(status_declared_wait_line "$state/$task.status") if [ -z "$last" ] || ! status_is_paused_or_captain_held "$last"; then reconcile_pause_tracking "$win" "$state" "$last" continue @@ -1180,7 +1181,7 @@ housekeeping() { # case "$?" in 2) rm -f "$marker" ;; *) - last=$(last_status_line "$state/$task.status") + last=$(status_declared_wait_line "$state/$task.status") if [ -n "$last" ] && status_is_captain_held "$last"; then if escalate_add "$state" "captain-held ${age}s (awaiting the captain, answer the held decision or release the hold): $win"; then _now > "$marker" @@ -1434,7 +1435,7 @@ handle_wake() { # pause) : ;; *) case "$stale_detail" in idle\ *s,\ possible\ wedge,\ escalation\ *) - last=$(last_status_line "$state/$task.status") + last=$(status_declared_wait_line "$state/$task.status") status_is_paused_or_captain_held "$last" \ || decision="escalate|${reason#stale: }" ;; @@ -1449,7 +1450,7 @@ handle_wake() { # [ "$kind" = signal ] && sync_pause_markers_from_signal "$state" "$arg" if [ "$kind" = stale ] && [ "$action" = escalate ]; then task=$(window_to_task "$arg" "$state") - last=$(last_status_line "$state/$task.status") + last=$(status_declared_wait_line "$state/$task.status") reconcile_pause_tracking "$arg" "$state" "$last" fi case "$action" in diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index d31a0adf8d5..4254f6fd4fb 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -1245,7 +1245,7 @@ wedge_wait_evidence() { # -> one wait_record on stdout local task=$1 last until statusf run [ -n "$task" ] || return 1 statusf="$STATE/$task.status" - last=$(last_status_line "$statusf") + last=$(status_declared_wait_line "$statusf") if status_is_captain_held "$last"; then wait_record 'captain-held' 'awaiting the captain - verified hold transfer' \ captain 'answer the held decision or release the hold' "$statusf" @@ -1543,7 +1543,7 @@ handle_paused_stale() { # case "$mtime" in ''|*[!0-9]*) mtime=$(date +%s) ;; esac now=$(date +%s) age=$(( now - mtime )) - last=$(last_status_line "$statusf") + last=$(status_declared_wait_line "$statusf") min_age=$PAUSE_RESURFACE_SECS declaration="declared:$(fm_wake_signal_sig "$statusf" || true)" if status_is_captain_held "$last"; then @@ -1601,7 +1601,7 @@ handle_paused_stale() { # busy_turn_bound_check() { # local win=$1 task=$2 h=$3 since_file=$4 escalation_file=$5 key statusf declared statusf="$STATE/$task.status" - if status_is_paused_or_captain_held "$(last_status_line "$statusf")"; then + if status_is_paused_or_captain_held "$(status_declared_wait_line "$statusf")"; then if afk_present; then # Away mode is daemon-owned, so this bound hands off the PLAIN wake identity # and lets the daemon classify the declaration itself - the undecorated @@ -1626,7 +1626,7 @@ busy_turn_bound_check() { # "$STATE/.stale-$key" triage_log "absorbed busy over-age pane (captain-held, never rechecked while the away-posture record exists): $win" return 0 @@ -1675,7 +1675,7 @@ clear_pause_tracking() { # pause_state_class() { # local win=$1 task=$2 key last recheck_file class agent_alive kind key=$(window_key "$win") - last=$(last_status_line "$STATE/$task.status") + last=$(status_declared_wait_line "$STATE/$task.status") recheck_file="$STATE/.paused-rechecked-$key" if ! status_is_paused_or_captain_held "$last"; then rm -f "$recheck_file" @@ -1848,7 +1848,7 @@ surface_nonterminal_stale() { # local win=$1 h=$2 key task last declared=1 bounded=1 throttled=1 until now key=$(window_key "$win") task=$(window_to_task "$win" "$STATE") - last=$(last_status_line "$STATE/$task.status") + last=$(status_declared_wait_line "$STATE/$task.status") STALE_WAIT_DECLARATION= if status_is_paused "$last"; then declared=0 @@ -2873,7 +2873,7 @@ EOF # exemption below, because a mate's steers land in an inbox too. [ -z "$task" ] || inbox_steer_check "$w" "$task" key=$(window_key "$w") - last=$(last_status_line "$STATE/$task.status") + last=$(status_declared_wait_line "$STATE/$task.status") if ! status_is_paused_or_captain_held "$last" && [ -e "$STATE/.paused-$key" ]; then clear_pause_tracking "$key" fi @@ -3016,7 +3016,7 @@ EOF esac else task=$(window_to_task "$w" "$STATE") - if [ -e "$pf" ] || status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")"; then + if [ -e "$pf" ] || status_is_paused_or_captain_held "$(status_declared_wait_line "$STATE/$task.status")"; then case "$(pause_state_class "$w" "$task")" in paused) handle_paused_stale "$w" "$task" "$h" ;; working) clear_pause_state "$key" @@ -3046,7 +3046,7 @@ EOF # is cleared - but not in the same poll the declared-pause cadence just # recorded it, or the re-surface throttle it depends on would be erased and # the pause would re-surface every poll instead of once per long cadence. - if [ "$paused_bound" -ne 0 ] && [ -e "$pf" ] && { [ "$n" -ge 2 ] || ! status_is_paused_or_captain_held "$(last_status_line "$STATE/$(window_to_task "$w" "$STATE").status")"; }; then + if [ "$paused_bound" -ne 0 ] && [ -e "$pf" ] && { [ "$n" -ge 2 ] || ! status_is_paused_or_captain_held "$(status_declared_wait_line "$STATE/$(window_to_task "$w" "$STATE").status")"; }; then clear_pause_tracking "$key" fi fi @@ -3061,7 +3061,7 @@ EOF clear_write_tracking "$key" fi task=$(window_to_task "$w" "$STATE") - if ! afk_present && status_is_paused_or_captain_held "$(last_status_line "$STATE/$task.status")" && [ "$busy_now" -ne 0 ]; then + if ! afk_present && status_is_paused_or_captain_held "$(status_declared_wait_line "$STATE/$task.status")" && [ "$busy_now" -ne 0 ]; then case "$(pause_state_class "$w" "$task")" in paused) handle_paused_stale "$w" "$task" "$h" ;; # Inconclusive, but the declared wait itself still stands, so only the diff --git a/docs/architecture.md b/docs/architecture.md index cb783557050..7cd1aa7a218 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -126,7 +126,7 @@ The most recent recognized ci log marker wins, so checks-green monitoring report `bin/fm-crew-state.sh` owns the evidence guard that recognizes ended CI monitors after green checks, including cancelled runs and skipped rebase steps; a passed run alone never proves a forge merge. In the coarse runs-ledger fallback, which has no steps table and no ci log, a terminal failed record whose daemon an explicit `daemon status` probe proves down reports unknown as unverified instead: an instrument failure must never read as work failure. The same instrument rule covers the ledger-anchored continuation of a selected run whose head this copy cannot resolve: once the probe answers down, that still-executing record reports unknown as unverified, while a run parked at a gate keeps its gate and findings because an open decision stays open when the instrument dies, and a `needs-decision` or `blocked` event the crew observed first hand stays open with the unverified record named as the reason rather than superseded by it. -Only when no matching run exists does it consult semantic busy state; exact busy reports working, exact idle permits fallback to the log's resolved current declaration - the newest decision the fold still holds open, otherwise the latest recognized event - when its verb maps to a recognized run-state, and unknown or a dead pane stays unknown instead of trusting a stale log. +Only when no matching run exists does it consult semantic busy state; exact busy reports working, exact idle permits fallback to the log's resolved current declaration - the newest decision the fold still holds open, otherwise a declared wait still standing after later resolved lines for other keys, otherwise the latest recognized event - when its verb maps to a recognized run-state, and unknown or a dead pane stays unknown instead of trusting a stale log. Decision-only events such as `resolved` never become current state or leak their prose into the current-state detail. In that status-log fallback, a declared external wait reports the distinct `paused` state with its reason. The semantic branch reports working only on an exact busy verdict and names the source that produced it; an unknown verdict never becomes working, never permits the status-log fallback, and never becomes a silent idle. @@ -192,6 +192,7 @@ A presence-gated sub-supervisor (`bin/fm-supervise-daemon.sh`) still extends wal The watcher and daemon share `bin/fm-classify-lib.sh` for captain-relevant status verbs, declared-wait vocabulary (a `paused:` external wait and a verified `captain-held` transfer alike, through one combined predicate), and status-scan primitives. Terminal verbs remain captain-relevant, while a nonterminal progress verb cannot become terminal merely because its prose contains a legacy free-text token such as `merged`; bare legacy free-text lines remain compatible. The shared latest-event read takes the most recent line that leads with a recognized verb or legacy token, so continuation prose and trailing blank lines after a multi-line record cannot hide a declared wait. +Both supervisors decide a declared wait through the library's declared-wait read rather than that latest event, so a later `resolved` line for a different phase key - including an `fm-send --resolve-key default` answer to a keyless decision - does not end a standing keyless or keyed `paused:` wait, while a resolved line for the wait's own key or any other later event still does. Both supervisors classify the status bytes appended since they last classified that log, never its last line alone, and report every actionable event through the captured endpoint before committing that position. The watcher's `.seen-*` and `.hb-surfaced-` markers and the daemon's `.subsuper-seen-status-` marker independently track reported file state and successfully classified position, so an unchanged unreadable state reports once without advancing past unread content, while a changed state retries and an unusable position re-reads the whole log. A keyed `needs-decision` or `blocked` transition accepted by the whole-file decision fold is retired only when that fold retires it - an explicit close for its exact key, or a terminal declaration by the ship or scout that owns the log - while a reserved-key transition the fold rejects surfaces as a reconciliation signal without becoming an open decision. diff --git a/tests/fm-classify-decision-key.test.sh b/tests/fm-classify-decision-key.test.sh index e6ede61d1e6..0419d24bce4 100755 --- a/tests/fm-classify-decision-key.test.sh +++ b/tests/fm-classify-decision-key.test.sh @@ -483,4 +483,81 @@ test_bare_prose_cannot_open_or_close_a_decision() { pass "only a colon-bearing or keyed line is a decision transition in the fold" } +# A stated [key=default] is the shared decision bucket --resolve-key default +# writes. It must keep closing a keyless decision, and it must not cancel a +# keyless live wait that only prints as default. A worker's own keyless +# resolved: still retracts that wait, and neither form closes a differently +# keyed wait. +test_keyless_wait_survives_stated_default_retraction() { + local dir f + dir=$(case_dir keyless-wait) + f="$dir/live.status" + printf 'needs-decision: which color\n' > "$f" + printf 'paused: waiting on the vendor release\n' >> "$f" + assert_fold "$f" "$(printf 'default\tneeds-decision\twhich color\n')" \ + "keyless decision stays open beside the wait" + [ "$(status_open_activities "$f")" = "$(printf 'default\tpaused\twaiting on the vendor release\n')" ] \ + || fail "keyless pause did not open as its own default phase: '$(status_open_activities "$f")'" + + printf 'resolved [key=default]: answered: blue\n' >> "$f" + assert_fold "$f" "" "stated default retraction closes the keyless decision" + [ "$(status_open_activities "$f")" = "$(printf 'default\tpaused\twaiting on the vendor release\n')" ] \ + || fail "stated default retraction cancelled the unrelated keyless wait: '$(status_open_activities "$f")'" + + printf 'paused: waiting on the vendor release\nresolved: [key=default] answered: blue\n' \ + > "$dir/colon-first.status" + [ "$(status_open_activities "$dir/colon-first.status")" = "$(printf 'default\tpaused\twaiting on the vendor release\n')" ] \ + || fail "a colon-first stated default retraction cancelled the keyless wait: '$(status_open_activities "$dir/colon-first.status")'" + + printf 'paused: waiting on the vendor release\n' > "$dir/self.status" + printf 'resolved: the vendor shipped\n' >> "$dir/self.status" + [ -z "$(status_open_activities "$dir/self.status")" ] \ + || fail "a keyless self-retraction left the keyless wait open: '$(status_open_activities "$dir/self.status")'" + + printf 'paused [key=legal]: awaiting counsel\n' > "$dir/keyed.status" + printf 'resolved [key=default]: answered: blue\n' >> "$dir/keyed.status" + printf 'resolved: unrelated keyless close\n' >> "$dir/keyed.status" + [ "$(status_open_activities "$dir/keyed.status")" = "$(printf 'legal\tpaused\tawaiting counsel\n')" ] \ + || fail "a default or keyless retraction closed a keyed wait: '$(status_open_activities "$dir/keyed.status")'" + + printf 'paused [key=default]: named default wait\n' > "$dir/stated.status" + printf 'resolved [key=default]: that wait cleared\n' >> "$dir/stated.status" + [ -z "$(status_open_activities "$dir/stated.status")" ] \ + || fail "a stated default retraction did not close the stated default wait" + + printf 'working: legacy start\ndone: legacy completion\n' > "$dir/legacy.status" + [ -z "$(status_open_activities "$dir/legacy.status")" ] \ + || fail "a keyless terminal stopped superseding the keyless working phase" + + printf 'paused: waiting on the vendor release\nneeds-decision [key=default]: which color\n' \ + > "$dir/stated-open.status" + [ "$(status_open_activities "$dir/stated-open.status")" = "$(printf 'default\tpaused\twaiting on the vendor release\n')" ] \ + || fail "a stated default decision cancelled the keyless wait: '$(status_open_activities "$dir/stated-open.status")'" + pass "a stated default retraction closes its decision and leaves an unrelated keyless wait standing" +} + +# The supervisors' declared-wait read keeps a pause standing behind answers +# for other keys even when those answers outrun the bounded tail window, and a +# resolved line for the pause's own key still retracts it from there. +test_declared_wait_survives_answers_past_the_event_window() { + local dir f i + dir=$(case_dir declared-wait-window) + f="$dir/answered.status" + printf 'needs-decision: which color\npaused: waiting on the vendor release\n' > "$f" + i=0 + while [ "$i" -le "$FM_CLASSIFY_EVENT_WINDOW_LINES" ]; do + printf 'resolved [key=q%s]: answered\n' "$i" >> "$f" + i=$((i + 1)) + done + printf 'resolved [key=default]: answered: blue\n' >> "$f" + [ "$(status_declared_wait_line "$f")" = 'paused: waiting on the vendor release' ] \ + || fail "answers past the event window cancelled the wait: '$(status_declared_wait_line "$f")'" + printf 'resolved: the vendor shipped\n' >> "$f" + [ -z "$(status_declared_wait_line "$f")" ] \ + || fail "the worker's own keyless resolved line did not retract the wait past the window" + pass "a declared wait outlives answers for other keys beyond the event window, and its own resolved line retracts it" +} + +test_keyless_wait_survives_stated_default_retraction +test_declared_wait_survives_answers_past_the_event_window test_bare_prose_cannot_open_or_close_a_decision diff --git a/tests/fm-daemon.test.sh b/tests/fm-daemon.test.sh index 74ec0905a10..21161cff276 100755 --- a/tests/fm-daemon.test.sh +++ b/tests/fm-daemon.test.sh @@ -907,6 +907,29 @@ test_stale_paused_classifies_pause() { pass "paused reasons with captain phrases remain pause-classified" } +# A resolved line for another phase key, including the stated default key that +# `fm-send --resolve-key default` writes for a keyless decision, lands after the +# pause without ending it. The worker's own keyless resolved line does end it. +test_stale_pause_survives_a_foreign_resolved_line() { + local dir state out + dir=$(make_supercase stale-paused-foreign-resolved) + state="$dir/state" + printf 'needs-decision: which color\npaused: waiting on the vendor release\nresolved [key=default]: answered: blue\n' \ + > "$state/held-w9r.status" + out=$(FM_STATE_OVERRIDE="$state" classify_stale "sess:fm-held-w9r" "$state" '' 1) + case "$out" in pause\|*"paused: waiting on the vendor release") ;; *) fail "a default-key answer cleared the pause: $out" ;; esac + printf 'paused: waiting on the vendor release\nresolved [key=legal]: counsel answered\n' > "$state/held-w9r.status" + out=$(FM_STATE_OVERRIDE="$state" classify_stale "sess:fm-held-w9r" "$state" '' 1) + case "$out" in pause\|*) ;; *) fail "a differently keyed resolved line cleared the pause: $out" ;; esac + printf 'paused: waiting on the vendor release\nresolved: the vendor shipped\n' > "$state/held-w9r.status" + out=$(FM_STATE_OVERRIDE="$state" classify_stale "sess:fm-held-w9r" "$state" '' 1) + case "$out" in pause\|*) fail "the worker's own keyless resolved line did not retract the pause: $out" ;; esac + printf 'captain-held [key=route]: tracked by task-decision-route\nresolved [key=default]: answered: blue\n' > "$state/held-w9r.status" + out=$(FM_STATE_OVERRIDE="$state" classify_stale "sess:fm-held-w9r" "$state" '' 1) + case "$out" in pause\|*) fail "a later resolved line no longer retracted a captain-held declaration: $out" ;; esac + pass "a foreign resolved line keeps a pause, while the worker's own resolved line retracts it" +} + # A verified captain-held transfer is the other declaration that leaves an idle pane # EXPECTED, so it earns the same pause action as paused: rather than being aged as a # wedge. The wait itself is already durable in the captain-held backlog task. @@ -2882,6 +2905,7 @@ test_enriched_wedge_under_declared_wait_uses_pause_cadence test_stale_terminal_escalates test_stale_actionable_wait_escalates_and_keeps_pause_cadence test_stale_paused_classifies_pause +test_stale_pause_survives_a_foreign_resolved_line test_stale_captain_held_classifies_pause test_handle_wake_paused_records_pause_marker test_handle_wake_paused_signal_records_pause_marker diff --git a/tests/fm-watch-triage.test.sh b/tests/fm-watch-triage.test.sh index fc31f44e616..8d44d1727e7 100755 --- a/tests/fm-watch-triage.test.sh +++ b/tests/fm-watch-triage.test.sh @@ -2891,6 +2891,40 @@ test_wedge_threshold_defers_to_a_declared_wait_under_a_working_verdict() { pass "a declared wait is not wedge-escalated by a working verdict, while an elapsed declaration and an undeclared lane both keep the unchanged ladder" } +# `fm-send --resolve-key default` answers a keyless decision by appending a +# stated default-key resolved line after whatever the worker wrote last. When +# that is a keyless pause the worker is still waiting, so the answer must not +# put the lane back on the wedge ladder. The worker's own keyless resolved line +# is the retraction that does. +test_wedge_threshold_keeps_a_wait_past_a_default_key_answer() { + local dir state fakebin out capture window key n + local working='state: working · source: run-step · ci running' + + dir=$(wedge_threshold_fixture default-answer-after-wait \ + "$(printf 'needs-decision: which color\npaused: waiting on the vendor release\nresolved [key=default]: answered: blue')" 0) + state="$dir/state"; fakebin="$dir/fakebin"; out="$dir/watch.out"; capture="$dir/pane.txt" + window="test:fm-wedge"; key=$(printf '%s' "$window" | tr ':/.' '___') + n=1 + while [ "$n" -le 3 ]; do + wedge_threshold_round "$state" "$fakebin" "$out" "$capture" "$window" "$working" absorb \ + || fail "a default-key answer put a waiting lane on the wedge ladder at threshold $n: $(cat "$out")" + n=$((n + 1)) + done + [ "$(wedge_stale_wakes "$state" "$window")" -eq 0 ] \ + || fail "a default-key answer let a waiting lane queue a wedge wake: $(cat "$state/.wake-queue")" + [ ! -e "$state/.wedge-escalations-$key" ] \ + || fail "a default-key answer let a waiting lane count $(cat "$state/.wedge-escalations-$key") wedge escalation(s)" + + dir=$(wedge_threshold_fixture keyless-retraction \ + "$(printf 'paused: waiting on the vendor release\nresolved: the vendor shipped')" 0) + state="$dir/state"; fakebin="$dir/fakebin"; out="$dir/watch.out"; capture="$dir/pane.txt" + wedge_threshold_round "$state" "$fakebin" "$out" "$capture" "$window" "$working" exit \ + || fail "a worker's own keyless resolved line did not retract its wait: $(cat "$out")" + grep -F "possible wedge, escalation 1" "$out" >/dev/null \ + || fail "a retracted wait did not return to the wedge ladder: $(cat "$out")" + pass "a default-key answer leaves a keyless wait standing, while the worker's own keyless resolved line retracts it" +} + # The other status-line record. A verified `captain-held:` transfer also reaches # this deferral - the mate has an active run attributed to it, so pause_state_class # reports working and the stable hash is handed to the wedge timer - but it blocks @@ -6218,6 +6252,7 @@ test_absorbed_replacement_wait_does_not_inherit_the_old_throttle test_live_declared_wait_churn_honors_the_resurface_throttle test_live_paused_until_controls_recheck_time test_wedge_threshold_defers_to_a_declared_wait_under_a_working_verdict +test_wedge_threshold_keeps_a_wait_past_a_default_key_answer test_wedge_threshold_recheck_names_the_captain_for_a_held_lane test_wedge_threshold_defers_to_a_parked_gate_awaiting_a_human test_wedge_threshold_parked_gate_needs_an_unanswered_decision From a8572f6255200c4809affb0d957be7889a9d33ca Mon Sep 17 00:00:00 2001 From: Tiago Date: Thu, 24 Sep 2026 23:21:35 -0300 Subject: [PATCH 04/84] fix(bin): terminate a remote job worker that lost ownership on TERM (#5544) * fix(bin): terminate a remote job worker that lost ownership when it receives TERM A serving worker whose lock directory is gone can no longer quarantine shutdown, and resuming service publishes a false ready heartbeat. Exit after stopping only that worker's own command tree, without removing a replacement owner's lock. * no-mistakes(review): Check worker lock ownership before publishing shutdown quarantine * no-mistakes(document): Correct worker shutdown comment on replacement-owned lock * fix(bin): keep an ousted remote job worker off the replacement quarantine Shutdown can lose the lock after the first ownership check and before it writes or clears quarantine. Bind both operations to the directory object this process still owns so a replacement's quarantine stays untouched. * no-mistakes(review): Make ousted-worker shutdown test reliably reach quarantine clear * no-mistakes(document): Reattach worker_shutdown doc comment to its function * no-mistakes(ci): Fixed the failing check (Behavior portable serial 7) with a test-only change to the stall test in tests/fm-remote-job.test.sh. Product code is unchanged; no other test changed. Cause: after the decoy dies, both workers run the same check-exists, read, delete sequence on the job records. On the CI runner the replacement deleted a record between the ousted worker's check and its read. The ousted worker exited 125, and because the file runs under set -e the unguarded `wait` ended the test with 125. The exit trap then killed the replacement, which produced the "Killed" line. Reproduction: a temporary 0.3 s delay between the check and the read, applied to the ousted worker only, made the committed test fail exactly as in CI (exit 125 and the "Killed" line). The new test passed with the same delay. The delay is reverted, along with a similar debug hook that the timed-out attempt had left in bin/fm-remote-job-worker.sh. Test changes: - The replacement is frozen (and confirmed stopped) before the decoy is killed and resumed only after the ousted worker exits, so only one worker touches the job records at a time. - The ousted worker is stopped only once its quarantine exists and its lane is reaped, which places it inside its stop loop. - Every fixed poll loop is now a wait on a named condition with a 30 s deadline and an explicit failure message. Exit detection also handles zombies. - The exit trap kills and waits for the decoy and both workers on every path. - A non-zero exit from the ousted worker now fails with its exit code and stderr instead of silently ending the file. The test still proves that the resumed ousted worker exits 0 and leaves the replacement's lock, quarantine contents and quarantine inode unchanged. Verification: the full test file passed four times on its own and three times under nice -n 10 with four busy-loop CPU hogs; bin/fm-lint.sh passes. Changes are not committed * no-mistakes(ci): I fixed the failing check (Behavior portable serial 7) by changing only the stall test in tests/fm-remote-job.test.sh. Product code is unchanged. **What failed:** "an ousted worker in shutdown leaves the replacement quarantine untouched" failed on CI with the ousted worker exiting 125 ("could not stop the active command tree"). **Why:** during shutdown, the worker retries the still-running decoy command group a fixed 100 times, 0.01 s apart, then gives up and exits 125. The test tried to freeze the worker partway through those retries by sending SIGSTOP from outside. On a slow runner the retries ran out before the stop arrived, so the worker had already given up. The invariant is that the test must hold the ousted worker inside that retry loop until the replacement owns the lock. That was the only place the test depended on timing. The other waits already watch for a named state change with a 30 s deadline. **Fix:** - The ousted worker now starts with a small `sleep` wrapper at the front of its PATH, and the SIGSTOP race is gone. - The wrapper only holds a `sleep` called directly by that worker's own process (it checks its parent pid against a hold file) while its quarantine file exists. - The only such `sleep` is the first retry in the shutdown stop loop, so the worker waits there as long as needed. - The wrapper writes a marker when it starts holding. The test waits for that marker, then hands the lock to the replacement, freezes the replacement, and kills the decoy. - The test releases the worker by deleting the hold file. Deleting the whole temp directory also releases it, so a failed run cannot leave the wrapper looping. - A process leak: the test overwrites the job's command-group record with the decoy, so no worker ever stopped the job's real command. `fm-hold-job.sh` and its `sleep 30` stayed running for up to 30 s after the test. The test now records that group before overwriting it and kills it at the end of the test and in the exit cleanup. - The test still asserts the same things: the ousted worker exits 0, and the replacement's lock, quarantine contents and quarantine inode are unchanged. **Verification:** - The full file passed twice on its own, twice under `nice -n 10` with six busy-loop CPU hogs, and twice more after the leak fix. - `pgrep` found no leftover processes afterwards. - With the worker from just before the fix commit (cf45cb6^), the test still fails with "the ousted worker wrote or cleared the replacement quarantine during shutdown", so it still proves the fix. - `bin/fm-lint.sh` passes. - I did not reproduce the CI failure locally. The cause comes from the fixed retry limit and the CI error message. The changes are not committed * no-mistakes(ci): I changed only the stall test ("an ousted worker in shutdown leaves the replacement quarantine untouched") in tests/fm-remote-job.test.sh. Product code is unchanged, and so is every other test. **Invariant:** the pid written to the job's group record must be a process-group leader whose group dies when that one process is killed. Otherwise the worker's bounded stop loop never sees the group die, gives up, and exits 125 ("could not stop the active command tree") before it reaches the lost-ownership exit. The decoy is the only place in this test that depends on this. **Fix:** - The decoy used to be `set -m; sleep 30 &`. It now starts as `perl -MPOSIX=setsid -e 'setsid() >= 0 or exit 1; exec @ARGV' sleep 30 &`, which gets its own session and group without shell job control. tests/fm-procevent.test.sh already uses the same idiom. - The test now waits, with the file's usual 30 s deadline and a named failure, until `ps -o pgid=` of the decoy equals its pid before writing it into the group record. This way the worker can never read the record before `setsid` has run. - The existing steps are unchanged: the test kills the decoy, reaps it with `wait` before releasing the hold file, and the exit trap still kills and reaps the decoy and both workers. - The assertions are unchanged: the ousted worker exits 0, and the replacement's lock pid, quarantine text and quarantine inode stay the same. **Cleanup:** I reverted a debug `printf` hook that the timed-out previous attempt had left in bin/fm-remote-job-worker.sh, and deleted its untracked `.tmp-repro/` directory. Neither was committed. **Verification:** - The full tests/fm-remote-job.test.sh passed twice normally and once under `setsid -w` with stdin from /dev/null (no controlling terminal). - `bin/fm-lint.sh` passes. - No leftover `sleep 30` processes afterwards. **Not reproduced:** I could not reproduce the CI failure locally. On this host `set -m` made the decoy its own group leader even without a controlling terminal, so the cause on the runner is not confirmed. The change removes the test's reliance on shell job control, as the user asked. Changes are not committed * no-mistakes(ci): I changed only the stall test ("an ousted worker in shutdown leaves the replacement quarantine untouched") in tests/fm-remote-job.test.sh. Product code is unchanged, and so is every other test. **Invariant:** the group record the ousted worker checks in its stop loop must stay the job's own command group, and the test must stop that group before it releases the hold. Otherwise the bounded retry keeps seeing a live group, gives up, and exits 125 ("could not stop the active command tree") before it reaches the lost-ownership exit. The test overwrote this record in one place (the decoy) and stopped the group in one place (killing the decoy); both are changed. **Fix:** - I removed the setsid decoy and the overwrite of `.claim/group`. The record keeps the job's real command group, which the test still saves as `STALL_JOB_GROUP`. - The two-line `group_start` stays. It is still needed: without it the worker kills the real group on its first pass, before the replacement takes over, so the hold would never matter. - The `sleep` wrapper that holds the worker at its first stop-loop retry is unchanged. - After the replacement owns the lock, its quarantine is planted and it is frozen, the test runs `kill -KILL -- -$STALL_JOB_GROUP`. It then waits, with the file's usual 30 s deadline and a named failure, until `kill -0` on the group fails. Only then does it remove the hold file. The worker therefore always sees its own command already stopped and never races its retry budget. - The exit trap still kills the saved command group if the test fails. It can't `wait` on that group because the group is not a child of the test shell. The decoy variable and its cleanup entry are gone. - The assertions are unchanged: the ousted worker exits 0, and the replacement's lock pid, quarantine text and quarantine inode stay the same. **Verification:** - The full tests/fm-remote-job.test.sh passed twice normally. - It passed once under `setsid -w` with stdin from /dev/null (no controlling terminal). - It passed once under `nice -n 10` with six busy-loop CPU hogs. - With the worker from before the fix (cf45cb6^), the test still fails with "the ousted worker wrote or cleared the replacement quarantine during shutdown", so it still proves the fix. - No `fm-hold-job` or `sleep 30` processes were left afterwards. - `bin/fm-lint.sh` passes. **Not reproduced:** I couldn't reproduce the CI failure locally; the decoy version also passed on this host. So I can't confirm why the decoy group stayed alive on the runner. The new wait turns any leftover live group into a clear named failure instead of an exit 125. The changes are not committed * fix(bin): keep a dead command group dead on bash 5.2 A bare return inside the liveness check drops the failing kill status when the check runs in a conditional, so shutdown keeps treating a stopped group as alive and exits 125. * no-mistakes(review): Use bash 3.2 fd syntax and fix trap return comments --- bin/fm-remote-job-worker.sh | 129 ++++++++++++-- tests/fm-remote-job.test.sh | 335 ++++++++++++++++++++++++++++++++++++ 2 files changed, 449 insertions(+), 15 deletions(-) diff --git a/bin/fm-remote-job-worker.sh b/bin/fm-remote-job-worker.sh index 14598eb7670..6d65c0ee44c 100755 --- a/bin/fm-remote-job-worker.sh +++ b/bin/fm-remote-job-worker.sh @@ -59,6 +59,7 @@ FM_ROOT=${FM_ROOT_OVERRIDE:-$(CDPATH='' cd "$SCRIPT_DIR/.." && pwd -P)} WORKER_LOCK= WORKER_LOCK_HELD=0 +WORKER_LOCK_BOUND= WORKER_RELEASE_OWNERSHIP=1 WORKER_SUPERVISED_PID= WORKER_PREEMPTIBLE=0 @@ -183,18 +184,73 @@ worker_acquire_lock() { return 1 } +# Open the lock directory this process still owns and remember a path that +# stays on that directory object. A replacement that removes the path and +# creates a new directory is invisible through a Linux directory fd, so a +# later write or clear cannot land in the replacement's quarantine. +worker_bind_owned_lock() { + local pid + [ "$WORKER_LOCK_HELD" -eq 1 ] || return 1 + [ -d "$WORKER_LOCK" ] && [ ! -L "$WORKER_LOCK" ] || return 1 + exec 9< "$WORKER_LOCK" || return 1 + if [ -d /proc/self/fd/9 ]; then + WORKER_LOCK_BOUND=/proc/self/fd/9 + else + WORKER_LOCK_BOUND=$WORKER_LOCK + fi + pid=$(fm_remote_job_read_single_line "$WORKER_LOCK_BOUND/pid" 64 2>/dev/null || true) + if [ "$pid" != "${BASHPID:-$$}" ]; then + worker_unbind_owned_lock + return 1 + fi +} + +worker_unbind_owned_lock() { + exec 9<&- + WORKER_LOCK_BOUND= +} + +worker_bound_lock_still_owned() { + local pid + [ -n "${WORKER_LOCK_BOUND:-}" ] || return 1 + pid=$(fm_remote_job_read_single_line "$WORKER_LOCK_BOUND/pid" 64 2>/dev/null || true) + [ "$pid" = "${BASHPID:-$$}" ] +} + worker_publish_quarantine() { local tmp - [ "$WORKER_LOCK_HELD" -eq 1 ] || return 1 - tmp=$(umask 077; mktemp "$WORKER_LOCK/.quarantine.XXXXXX") || return 1 - printf 'active execution could not be confirmed stopped\n' > "$tmp" || { rm -f -- "$tmp"; return 1; } - chmod 600 "$tmp" || { rm -f -- "$tmp"; return 1; } - mv -f -- "$tmp" "$WORKER_LOCK/quarantine" + worker_bind_owned_lock || return 1 + tmp=$(umask 077; mktemp "$WORKER_LOCK_BOUND/.quarantine.XXXXXX") || { worker_unbind_owned_lock; return 1; } + if ! printf 'active execution could not be confirmed stopped\n' > "$tmp" \ + || ! chmod 600 "$tmp" || ! worker_bound_lock_still_owned \ + || ! mv -f -- "$tmp" "$WORKER_LOCK_BOUND/quarantine"; then + rm -f -- "$tmp" + worker_unbind_owned_lock + return 1 + fi + worker_unbind_owned_lock } worker_clear_quarantine() { - [ ! -L "$WORKER_LOCK/quarantine" ] || return 1 - rm -f -- "$WORKER_LOCK/quarantine" + worker_bind_owned_lock || return 1 + if [ -L "$WORKER_LOCK_BOUND/quarantine" ] || ! worker_bound_lock_still_owned \ + || ! rm -f -- "$WORKER_LOCK_BOUND/quarantine"; then + worker_unbind_owned_lock + return 1 + fi + worker_unbind_owned_lock +} + +# True only while this process still owns the lock directory it published. +# A missing directory, or a directory whose pid is not this process, belongs +# to a replacement or to nobody. Shutdown must not remove it or signal work +# recorded only under that replacement. +worker_shutdown_owns_lock() { + local owner_pid + [ "$WORKER_LOCK_HELD" -eq 1 ] || return 1 + [ -d "$WORKER_LOCK" ] && [ ! -L "$WORKER_LOCK" ] || return 1 + owner_pid=$(fm_remote_job_read_single_line "$WORKER_LOCK/pid" 64 2>/dev/null || true) + [ "$owner_pid" = "${BASHPID:-$$}" ] } worker_cleanup() { @@ -296,7 +352,13 @@ worker_recorded_execution_alive() { # process|group case "$identity_status" in 0) ;; 1) return 1 ;; - 2) worker_process_or_group_alive process "$pid"; return ;; + 2) + # This runs inside the shutdown and exit traps, where a bare return + # reports the status from before the trap, so a dead process would + # still look alive. + worker_process_or_group_alive process "$pid" + return $? + ;; esac else worker_group_identity_status "$job" "$pid" @@ -304,7 +366,13 @@ worker_recorded_execution_alive() { # process|group case "$identity_status" in 0|3) ;; 1) return 1 ;; - 2) worker_process_or_group_alive group "$pid"; return ;; + 2) + # This runs inside the shutdown and exit traps, where a bare return + # reports the status from before the trap, so a dead group would + # still look alive. + worker_process_or_group_alive group "$pid" + return $? + ;; esac fi worker_process_or_group_alive "$kind" "$pid" @@ -388,6 +456,18 @@ worker_stop_active_execution() { [ "$failed" -eq 0 ] } +# Ownership is already gone. Stop only this process's command tree and exit +# without releasing or rewriting the directory a replacement may now own. +worker_exit_lost_lock() { + WORKER_RELEASE_OWNERSHIP=0 + WORKER_LOCK_HELD=0 + worker_stop_active_execution || { + worker_error "could not stop the active command tree" + exit 125 + } + exit 0 +} + # Ignore, rather than restore the default disposition for, the signals this # handler answers. A replacement stops a Linux worker by signalling its whole # isolated group, and the supervisor in that group forwards a second stop signal @@ -399,10 +479,26 @@ worker_stop_active_execution() { # KILL, which no disposition can block. worker_shutdown() { trap '' HUP INT TERM + # The ownership directory is gone or a replacement owns it. TERM stays + # authoritative: stop only this process's command tree, then exit without + # touching the directory, whose files, quarantine included, now belong to + # the replacement or to nobody. Drop the in-memory hold first so exit + # cleanup cannot release a replacement's lock. Signals stay ignored until + # exit, so a repeat is a no-op. + if ! worker_shutdown_owns_lock; then + worker_exit_lost_lock + fi + # Still our lock: a transient publish failure must not abandon the + # directory. Re-arm and keep serving so a later signal can quarantine it. + # A publish failure after the directory was replaced is lost ownership, + # not a reason to keep serving. worker_publish_quarantine || { - worker_error "cannot guard worker ownership for shutdown" - trap worker_shutdown HUP INT TERM - return 0 + if worker_shutdown_owns_lock; then + worker_error "cannot guard worker ownership for shutdown" + trap worker_shutdown HUP INT TERM + return 0 + fi + worker_exit_lost_lock } worker_stop_active_execution || { worker_error "could not stop the active command tree" @@ -410,9 +506,12 @@ worker_shutdown() { exit 125 } worker_clear_quarantine || { - worker_error "could not clear guarded worker ownership after shutdown" - WORKER_RELEASE_OWNERSHIP=0 - exit 125 + if worker_shutdown_owns_lock; then + worker_error "could not clear guarded worker ownership after shutdown" + WORKER_RELEASE_OWNERSHIP=0 + exit 125 + fi + worker_exit_lost_lock } exit 0 } diff --git a/tests/fm-remote-job.test.sh b/tests/fm-remote-job.test.sh index 61c8bb8d149..19023e04cd9 100755 --- a/tests/fm-remote-job.test.sh +++ b/tests/fm-remote-job.test.sh @@ -20,6 +20,11 @@ OTHER_PID= RECOVERY_WORKER_PID= REPEAT_WORKER_PID= RESTART_SUPERVISOR_PID= +LOST_TERM_PID= +REPLACEMENT_OWNER_PID= +STALL_WORKER_PID= +STALL_REPLACEMENT_PID= +STALL_JOB_GROUP= mkdir -p "$REMOTE_ROOT/bin" "$REMOTE_HOME" "$ACCOUNT_HOME" "$RUNTIME_BIN" # worker.pid records the serving child, not its restart supervisor, so stopping # that pid alone leaves the supervisor to respawn - the leak @@ -29,6 +34,15 @@ cleanup_remote_job_fixture() { [ -z "$RECOVERY_WORKER_PID" ] || kill "$RECOVERY_WORKER_PID" 2>/dev/null || true [ -z "$REPEAT_WORKER_PID" ] || kill "$REPEAT_WORKER_PID" 2>/dev/null || true [ -z "$RESTART_SUPERVISOR_PID" ] || kill -KILL "$RESTART_SUPERVISOR_PID" 2>/dev/null || true + [ -z "$LOST_TERM_PID" ] || kill -KILL "$LOST_TERM_PID" 2>/dev/null || true + [ -z "$REPLACEMENT_OWNER_PID" ] || kill -KILL "$REPLACEMENT_OWNER_PID" 2>/dev/null || true + local stall_pid + for stall_pid in "$STALL_WORKER_PID" "$STALL_REPLACEMENT_PID"; do + [ -n "$stall_pid" ] || continue + kill -KILL "$stall_pid" 2>/dev/null || true + wait "$stall_pid" 2>/dev/null || true + done + [ -z "$STALL_JOB_GROUP" ] || kill -KILL -- "-$STALL_JOB_GROUP" 2>/dev/null || true if [ -f "$STATE_ROOT/worker.pid" ]; then fm_remote_job_stop_worker_tree "$(cat "$STATE_ROOT/worker.pid")" || true fi @@ -204,6 +218,14 @@ file_mode() { fi } +file_inode() { + if [ "$(uname)" = Darwin ]; then + stat -f %i "$1" 2>/dev/null || true + else + stat -c %i "$1" 2>/dev/null || true + fi +} + printf 'first line\nsecond line\n' > "$TMP_ROOT/stdin" # shellcheck disable=SC2016 # Literal shell-looking argv is an injection probe. TOP_SECRET=must-not-cross fm_remote_job_stage "$ACCOUNT_HOME" "$REMOTE_ROOT" "$REMOTE_HOME" \ @@ -716,6 +738,319 @@ wait "$REPEAT_WORKER_PID" 2>/dev/null || true REPEAT_WORKER_PID= pass "a repeatedly signalled shutdown still releases ownership for the next worker" +cat > "$REMOTE_ROOT/bin/fm-hold-job.sh" <<'SH' +#!/bin/bash +trap '' HUP INT TERM +printf 'started\n' > "$1" +sleep 30 +printf 'ran\n' > "$2" +SH +chmod +x "$REMOTE_ROOT/bin/fm-hold-job.sh" +git -C "$REMOTE_ROOT" add bin/fm-hold-job.sh +git -C "$REMOTE_ROOT" commit -qm 'hold job' + +LOST_HOME="$TMP_ROOT/lost-term-account" +LOST_STATE="$TMP_ROOT/lost-term-jobs" +mkdir -p "$LOST_HOME" +chmod 700 "$LOST_HOME" +HOME="$LOST_HOME" FM_ROOT_OVERRIDE="$REMOTE_ROOT" FM_REMOTE_JOB_STATE_ROOT="$LOST_STATE" \ + FM_REMOTE_JOB_PLATFORM_OVERRIDE=Linux "$REMOTE_ROOT/bin/fm-remote-job-worker.sh" --serve \ + > "$TMP_ROOT/lost-term.out" 2> "$TMP_ROOT/lost-term.err" & +LOST_TERM_PID=$! +for _ in $(seq 1 300); do + [ -f "$LOST_STATE/worker.ready" ] && break + sleep 0.05 +done +assert_present "$LOST_STATE/worker.ready" "the ownership-loss worker did not become ready" +assert_present "$LOST_STATE/worker.lock" "the ownership-loss worker did not publish its lock" +kill -STOP "$LOST_TERM_PID" +for _ in $(seq 1 100); do + [ "$(ps -o state= -p "$LOST_TERM_PID" 2>/dev/null | tr -d ' ')" = T ] && break + sleep 0.05 +done +[ "$(ps -o state= -p "$LOST_TERM_PID" 2>/dev/null | tr -d ' ')" = T ] \ + || fail "the ownership-loss worker did not stop" +rm -rf -- "$LOST_STATE/worker.lock" +kill -CONT "$LOST_TERM_PID" +LOST_READY_BEFORE=$(file_inode "$LOST_STATE/worker.ready") +for _ in $(seq 1 100); do + LOST_READY_AFTER=$(file_inode "$LOST_STATE/worker.ready") + [ -n "$LOST_READY_AFTER" ] && [ "$LOST_READY_AFTER" != "$LOST_READY_BEFORE" ] && break + sleep 0.05 +done +[ -n "${LOST_READY_AFTER:-}" ] && [ "$LOST_READY_AFTER" != "$LOST_READY_BEFORE" ] \ + || fail "a worker with no ownership lock stopped publishing heartbeats before TERM" +assert_absent "$LOST_STATE/worker.lock" "the ownership lock reappeared before TERM" +kill -TERM "$LOST_TERM_PID" +for _ in $(seq 1 100); do + kill -0 "$LOST_TERM_PID" 2>/dev/null || break + sleep 0.05 +done +if kill -0 "$LOST_TERM_PID" 2>/dev/null; then + fail "TERM after ownership loss left the serving worker alive" +fi +wait "$LOST_TERM_PID" 2>/dev/null || true +LOST_TERM_PID= +LOST_READY_SETTLED=$(file_inode "$LOST_STATE/worker.ready") +sleep 0.3 +[ "$(file_inode "$LOST_STATE/worker.ready")" = "$LOST_READY_SETTLED" ] \ + || fail "a worker that lost ownership kept replacing its heartbeat after TERM" +pass "TERM after ownership loss stops the serving worker" + +HOLD_STARTED="$TMP_ROOT/hold-started" +HOLD_SIDE_EFFECT="$TMP_ROOT/hold-side-effect" +rm -f -- "$HOLD_STARTED" "$HOLD_SIDE_EFFECT" +LOST_HOME="$TMP_ROOT/lost-cleanup-account" +LOST_STATE="$TMP_ROOT/lost-cleanup-jobs" +mkdir -p "$LOST_HOME" +chmod 700 "$LOST_HOME" +HOME="$LOST_HOME" FM_ROOT_OVERRIDE="$REMOTE_ROOT" FM_REMOTE_JOB_STATE_ROOT="$LOST_STATE" \ + FM_REMOTE_JOB_PLATFORM_OVERRIDE=Linux "$REMOTE_ROOT/bin/fm-remote-job-worker.sh" --serve \ + > "$TMP_ROOT/lost-cleanup.out" 2> "$TMP_ROOT/lost-cleanup.err" & +LOST_TERM_PID=$! +for _ in $(seq 1 300); do + [ -f "$LOST_STATE/worker.ready" ] && break + sleep 0.05 +done +assert_present "$LOST_STATE/worker.ready" "the cleanup worker did not become ready" +FM_REMOTE_JOB_STATE_ROOT="$LOST_STATE" FM_REMOTE_JOB_TIMEOUT=20 \ + fm_remote_job_stage "$LOST_HOME" "$REMOTE_ROOT" "$REMOTE_HOME" \ + fm-hold-job.sh "$HOLD_STARTED" "$HOLD_SIDE_EFFECT" < /dev/null > /dev/null +for _ in $(seq 1 100); do + [ -f "$HOLD_STARTED" ] && break + sleep 0.05 +done +assert_present "$HOLD_STARTED" "the held command did not start before ownership loss" +kill -STOP "$LOST_TERM_PID" +for _ in $(seq 1 100); do + [ "$(ps -o state= -p "$LOST_TERM_PID" 2>/dev/null | tr -d ' ')" = T ] && break + sleep 0.05 +done +rm -rf -- "$LOST_STATE/worker.lock" +kill -TERM "$LOST_TERM_PID" +kill -CONT "$LOST_TERM_PID" +for _ in $(seq 1 100); do + kill -0 "$LOST_TERM_PID" 2>/dev/null || break + sleep 0.05 +done +if kill -0 "$LOST_TERM_PID" 2>/dev/null; then + fail "TERM after ownership loss did not stop a worker with an active command" +fi +wait "$LOST_TERM_PID" 2>/dev/null || true +LOST_TERM_PID= +sleep 0.5 +assert_absent "$HOLD_SIDE_EFFECT" "the active command kept running after an unowned TERM" +pass "TERM after ownership loss still stops the active command tree" + +OWNER_HOME="$TMP_ROOT/replacement-owner-account" +OWNER_STATE="$TMP_ROOT/replacement-owner-jobs" +OWNER_STARTED="$TMP_ROOT/replacement-started" +OWNER_SIDE_EFFECT="$TMP_ROOT/replacement-side-effect" +mkdir -p "$OWNER_HOME" +chmod 700 "$OWNER_HOME" +HOME="$OWNER_HOME" FM_ROOT_OVERRIDE="$REMOTE_ROOT" FM_REMOTE_JOB_STATE_ROOT="$OWNER_STATE" \ + FM_REMOTE_JOB_PLATFORM_OVERRIDE=Linux "$REMOTE_ROOT/bin/fm-remote-job-worker.sh" --serve \ + > "$TMP_ROOT/replacement-lost.out" 2> "$TMP_ROOT/replacement-lost.err" & +LOST_TERM_PID=$! +for _ in $(seq 1 300); do + [ -f "$OWNER_STATE/worker.ready" ] && break + sleep 0.05 +done +assert_present "$OWNER_STATE/worker.ready" "the worker that will lose ownership did not become ready" +kill -STOP "$LOST_TERM_PID" +for _ in $(seq 1 100); do + [ "$(ps -o state= -p "$LOST_TERM_PID" 2>/dev/null | tr -d ' ')" = T ] && break + sleep 0.05 +done +rm -rf -- "$OWNER_STATE/worker.lock" +HOME="$OWNER_HOME" FM_ROOT_OVERRIDE="$REMOTE_ROOT" FM_REMOTE_JOB_STATE_ROOT="$OWNER_STATE" \ + FM_REMOTE_JOB_PLATFORM_OVERRIDE=Linux "$REMOTE_ROOT/bin/fm-remote-job-worker.sh" --serve \ + > "$TMP_ROOT/replacement-owner.out" 2> "$TMP_ROOT/replacement-owner.err" & +REPLACEMENT_OWNER_PID=$! +for _ in $(seq 1 300); do + [ -f "$OWNER_STATE/worker.lock/pid" ] && [ "$(cat "$OWNER_STATE/worker.lock/pid")" = "$REPLACEMENT_OWNER_PID" ] && break + sleep 0.05 +done +[ "$(cat "$OWNER_STATE/worker.lock/pid" 2>/dev/null || true)" = "$REPLACEMENT_OWNER_PID" ] \ + || fail "the replacement worker did not take ownership" +FM_REMOTE_JOB_STATE_ROOT="$OWNER_STATE" FM_REMOTE_JOB_TIMEOUT=20 \ + fm_remote_job_stage "$OWNER_HOME" "$REMOTE_ROOT" "$REMOTE_HOME" \ + fm-hold-job.sh "$OWNER_STARTED" "$OWNER_SIDE_EFFECT" < /dev/null > /dev/null +for _ in $(seq 1 100); do + [ -f "$OWNER_STARTED" ] && break + sleep 0.05 +done +assert_present "$OWNER_STARTED" "the replacement worker's command did not start" +OWNER_JOB_SUPERVISOR=$(cat "$OWNER_STATE/jobs/$FM_REMOTE_JOB_ID/.claim/supervisor") +printf 'replacement guard\n' > "$OWNER_STATE/worker.lock/quarantine" +OWNER_QUARANTINE_INODE=$(file_inode "$OWNER_STATE/worker.lock/quarantine") +LATE_BURST=0 +while [ "$LATE_BURST" -lt 10 ]; do + kill -TERM "$LOST_TERM_PID" 2>/dev/null || true + LATE_BURST=$((LATE_BURST + 1)) +done +kill -CONT "$LOST_TERM_PID" +for _ in $(seq 1 100); do + kill -0 "$LOST_TERM_PID" 2>/dev/null || break + sleep 0.05 +done +if kill -0 "$LOST_TERM_PID" 2>/dev/null; then + fail "a burst of TERMs after ownership loss left the old worker alive" +fi +wait "$LOST_TERM_PID" 2>/dev/null || true +LOST_DEAD_PID=$LOST_TERM_PID +LOST_TERM_PID= +kill -TERM "$LOST_DEAD_PID" 2>/dev/null || true +kill -0 "$REPLACEMENT_OWNER_PID" 2>/dev/null \ + || fail "terminating the old worker also terminated the replacement owner" +kill -0 "$OWNER_JOB_SUPERVISOR" 2>/dev/null \ + || fail "terminating the old worker stopped the replacement owner's command" +[ "$(cat "$OWNER_STATE/worker.lock/pid" 2>/dev/null || true)" = "$REPLACEMENT_OWNER_PID" ] \ + || fail "the old worker's cleanup removed the replacement owner's lock" +[ "$(cat "$OWNER_STATE/worker.lock/quarantine" 2>/dev/null || true)" = "replacement guard" ] \ + && [ "$(file_inode "$OWNER_STATE/worker.lock/quarantine")" = "$OWNER_QUARANTINE_INODE" ] \ + || fail "the old worker's TERM rewrote or removed the replacement owner's quarantine" +rm -f -- "$OWNER_STATE/worker.lock/quarantine" +assert_absent "$OWNER_SIDE_EFFECT" "the replacement command finished during the ownership handoff" +kill -TERM "$REPLACEMENT_OWNER_PID" +for _ in $(seq 1 100); do + kill -0 "$REPLACEMENT_OWNER_PID" 2>/dev/null || break + sleep 0.05 +done +if kill -0 "$REPLACEMENT_OWNER_PID" 2>/dev/null; then + fail "the replacement owner did not finish its own TERM shutdown" +fi +wait "$REPLACEMENT_OWNER_PID" 2>/dev/null || true +REPLACEMENT_OWNER_PID= +assert_absent "$OWNER_STATE/worker.lock" \ + "the replacement owner's shutdown left its lock behind" +sleep 0.5 +assert_absent "$OWNER_SIDE_EFFECT" \ + "the replacement owner's command kept running after its own shutdown" +pass "a lost owner terminates without stopping the replacement owner's work" + +# Shutdown publishes quarantine, then stops the command, then clears quarantine. +# Steal the lock in that gap: the ousted worker must not write or clear the +# replacement's quarantine when it resumes. +STALL_HOME="$TMP_ROOT/stall-owner-account" +STALL_STATE="$TMP_ROOT/stall-owner-jobs" +STALL_STARTED="$TMP_ROOT/stall-started" +STALL_SIDE_EFFECT="$TMP_ROOT/stall-side-effect" +STALL_BIN="$TMP_ROOT/stall-bin" +STALL_HOLD="$TMP_ROOT/stall-hold" +STALL_HELD="$TMP_ROOT/stall-held" +mkdir -p "$STALL_HOME" "$STALL_BIN" +chmod 700 "$STALL_HOME" +# The stop loop gives up after a bounded number of retries, so pausing the +# worker from outside races that bound on a slow runner. This sleep holds the +# worker's own shell at its first stop-loop retry instead: that is the only +# sleep it runs while its quarantine exists. Removing the hold file (or the +# whole fixture) releases it. +cat > "$STALL_BIN/sleep" < '$STALL_HELD' + while [ -e '$STALL_HOLD' ]; do '$(command -v sleep)' 0.05; done +fi +exec '$(command -v sleep)' "\$@" +SH +chmod +x "$STALL_BIN/sleep" +HOME="$STALL_HOME" FM_ROOT_OVERRIDE="$REMOTE_ROOT" FM_REMOTE_JOB_STATE_ROOT="$STALL_STATE" \ + FM_REMOTE_JOB_PLATFORM_OVERRIDE=Linux PATH="$STALL_BIN:$PATH" \ + "$REMOTE_ROOT/bin/fm-remote-job-worker.sh" --serve \ + > "$TMP_ROOT/stall-lost.out" 2> "$TMP_ROOT/stall-lost.err" & +STALL_WORKER_PID=$! +printf '%s\n' "$STALL_WORKER_PID" > "$STALL_HOLD" +STALL_DEADLINE=$((SECONDS + 30)) +until [ -f "$STALL_STATE/worker.ready" ] || [ "$SECONDS" -ge "$STALL_DEADLINE" ]; do sleep 0.05; done +assert_present "$STALL_STATE/worker.ready" "the worker stalled in shutdown did not become ready" +FM_REMOTE_JOB_STATE_ROOT="$STALL_STATE" FM_REMOTE_JOB_TIMEOUT=20 \ + fm_remote_job_stage "$STALL_HOME" "$REMOTE_ROOT" "$REMOTE_HOME" \ + fm-hold-job.sh "$STALL_STARTED" "$STALL_SIDE_EFFECT" < /dev/null > /dev/null +STALL_DEADLINE=$((SECONDS + 30)) +until [ -f "$STALL_STARTED" ] || [ "$SECONDS" -ge "$STALL_DEADLINE" ]; do sleep 0.05; done +assert_present "$STALL_STARTED" "the command that keeps shutdown in its stop loop did not start" +STALL_JOB="$STALL_STATE/jobs/$FM_REMOTE_JOB_ID" +# An unreadable start record keeps the worker from signalling the job's own +# command group while it still counts that group as alive, so shutdown waits in +# its stop loop until the test stops the group. +STALL_JOB_GROUP=$(cat "$STALL_JOB/.claim/group") +printf 'unconfirmed\nstart\n' > "$STALL_JOB/.claim/group_start" +kill -TERM "$STALL_WORKER_PID" +STALL_DEADLINE=$((SECONDS + 30)) +until [ -f "$STALL_HELD" ] || [ "$SECONDS" -ge "$STALL_DEADLINE" ]; do sleep 0.05; done +assert_present "$STALL_HELD" "shutdown did not reach its stop loop behind its own quarantine" +assert_present "$STALL_STATE/worker.lock/quarantine" \ + "the worker held in its stop loop did not hold its own quarantine" +rm -rf -- "$STALL_STATE/worker.lock" +HOME="$STALL_HOME" FM_ROOT_OVERRIDE="$REMOTE_ROOT" FM_REMOTE_JOB_STATE_ROOT="$STALL_STATE" \ + FM_REMOTE_JOB_PLATFORM_OVERRIDE=Linux "$REMOTE_ROOT/bin/fm-remote-job-worker.sh" --serve \ + > "$TMP_ROOT/stall-replacement.out" 2> "$TMP_ROOT/stall-replacement.err" & +STALL_REPLACEMENT_PID=$! +STALL_DEADLINE=$((SECONDS + 30)) +until [ "$(cat "$STALL_STATE/worker.lock/pid" 2>/dev/null || true)" = "$STALL_REPLACEMENT_PID" ] \ + || [ "$SECONDS" -ge "$STALL_DEADLINE" ]; do + sleep 0.05 +done +[ "$(cat "$STALL_STATE/worker.lock/pid" 2>/dev/null || true)" = "$STALL_REPLACEMENT_PID" ] \ + || fail "the replacement did not take ownership while the old worker was stopped in shutdown" +printf 'replacement guard\n' > "$STALL_STATE/worker.lock/quarantine" +STALL_QUARANTINE_INODE=$(file_inode "$STALL_STATE/worker.lock/quarantine") +# Both workers stop the job's recorded execution and then delete its records. +# A worker resumed while the other is deleting can lose a record read and exit +# before its quarantine clear, so freeze the replacement until the ousted +# worker has finished. +kill -STOP "$STALL_REPLACEMENT_PID" +STALL_DEADLINE=$((SECONDS + 30)) +until [ "$(ps -o state= -p "$STALL_REPLACEMENT_PID" 2>/dev/null | tr -d ' ')" = T ] \ + || [ "$SECONDS" -ge "$STALL_DEADLINE" ]; do + sleep 0.05 +done +[ "$(ps -o state= -p "$STALL_REPLACEMENT_PID" 2>/dev/null | tr -d ' ')" = T ] \ + || fail "the replacement could not be held while the ousted worker resumed" +kill -KILL -- "-$STALL_JOB_GROUP" 2>/dev/null || true +STALL_DEADLINE=$((SECONDS + 30)) +until ! kill -0 -- "-$STALL_JOB_GROUP" 2>/dev/null || [ "$SECONDS" -ge "$STALL_DEADLINE" ]; do + sleep 0.05 +done +! kill -0 -- "-$STALL_JOB_GROUP" 2>/dev/null \ + || fail "the job's command group was still alive after the test stopped it" +rm -f -- "$STALL_HOLD" +STALL_DEADLINE=$((SECONDS + 30)) +until [ "$(ps -o state= -p "$STALL_WORKER_PID" 2>/dev/null | tr -d ' ')" = Z ] \ + || ! kill -0 "$STALL_WORKER_PID" 2>/dev/null || [ "$SECONDS" -ge "$STALL_DEADLINE" ]; do + sleep 0.05 +done +[ "$(ps -o state= -p "$STALL_WORKER_PID" 2>/dev/null | tr -d ' ')" = Z ] \ + || ! kill -0 "$STALL_WORKER_PID" 2>/dev/null \ + || fail "the ousted worker did not exit after shutdown resumed" +STALL_WORKER_RC=0 +wait "$STALL_WORKER_PID" 2>/dev/null || STALL_WORKER_RC=$? +STALL_WORKER_PID= +[ "$STALL_WORKER_RC" -eq 0 ] \ + || fail "the ousted worker did not finish shutdown through its lost-ownership exit (exit $STALL_WORKER_RC: $(cat "$TMP_ROOT/stall-lost.err"))" +kill -CONT "$STALL_REPLACEMENT_PID" +kill -0 "$STALL_REPLACEMENT_PID" 2>/dev/null \ + || fail "the ousted worker's resumed shutdown terminated the replacement" +[ "$(cat "$STALL_STATE/worker.lock/pid" 2>/dev/null || true)" = "$STALL_REPLACEMENT_PID" ] \ + || fail "the ousted worker's resumed shutdown removed the replacement lock" +[ "$(cat "$STALL_STATE/worker.lock/quarantine" 2>/dev/null || true)" = "replacement guard" ] \ + && [ "$(file_inode "$STALL_STATE/worker.lock/quarantine")" = "$STALL_QUARANTINE_INODE" ] \ + || fail "the ousted worker wrote or cleared the replacement quarantine during shutdown" +kill -TERM "$STALL_REPLACEMENT_PID" +STALL_DEADLINE=$((SECONDS + 30)) +until [ "$(ps -o state= -p "$STALL_REPLACEMENT_PID" 2>/dev/null | tr -d ' ')" = Z ] \ + || ! kill -0 "$STALL_REPLACEMENT_PID" 2>/dev/null || [ "$SECONDS" -ge "$STALL_DEADLINE" ]; do + sleep 0.05 +done +[ "$(ps -o state= -p "$STALL_REPLACEMENT_PID" 2>/dev/null | tr -d ' ')" = Z ] \ + || ! kill -0 "$STALL_REPLACEMENT_PID" 2>/dev/null \ + || fail "the replacement did not finish its own TERM shutdown" +wait "$STALL_REPLACEMENT_PID" 2>/dev/null || true +STALL_REPLACEMENT_PID= +STALL_JOB_GROUP= +pass "an ousted worker in shutdown leaves the replacement quarantine untouched" + # A child that stays up for FM_REMOTE_JOB_SUPERVISOR_HEALTHY_SECONDS clears the # consecutive-failure backoff, so a child that dies just past that threshold # used to reset the only guard the supervisor had and restart forever. The From 4be8a409597572ad2e29bb6186dc71d4ec3bb787 Mon Sep 17 00:00:00 2001 From: Trevin Chow Date: Thu, 24 Sep 2026 21:24:49 -0700 Subject: [PATCH 05/84] docs: make configuration settings easier to find and understand (#5589) * docs: make configuration settings easier to find and understand * no-mistakes(review): Restore dropped qualifiers and fix misplaced config doc labels * no-mistakes(review): Restore three dropped qualifiers in configuration reference --- docs/configuration.md | 1691 +++++++++++++++++++++++++++++++++-------- 1 file changed, 1363 insertions(+), 328 deletions(-) diff --git a/docs/configuration.md b/docs/configuration.md index 30f91710442..e96a7b2c752 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -1,107 +1,318 @@ # Configuration -The files and environment variables you set to operate firstmate. +Configure where Firstmate keeps its files, which tools launch workers, and how supervision runs. +Start with the directory layout, then use the setting reference for the behavior you want to change. -## Orchestrator behavior (AGENTS.md) +## Find a setting + +| What you want to configure | Start here | +| --- | --- | +| Firstmate's code, private files, or project location | [FM_HOME](#fm_home) and [operational home layout](#operational-home-layout-and-state) | +| Task windows and worker tools | [Runtime backend](#runtime-backend-configbackend--fm_backend) and [harness support](#harness-support) | +| Worker permissions, accounts, or environment | [Claude permission mode](#claude-permission-mode-configclaude-permission-mode), [worker account pin](#worker-account-pin-configclaude-account-configpi-account), and [worker launch environment](#worker-launch-environment-configlaunch-env-allowlist) | +| Backlog, preferences, and memory | [Backlog backend](#backlog-backend-taskstoml--configbacklog-backend), [captain preferences](#captain-preferences-datacaptainmd--datacaptain-sharedmd), and [startup memory budget](#startup-memory-budget-configstartup-memory-budget) | +| Supervision and presentation | [Pi supervision branch](#pi-supervision-branch), [supervision host](#supervision-host-configsupervision-host), and [Calm preference](#calm-preference-configcalm) | +| Persistent secondmates | [Secondmate routes](#secondmate-routes-datasecondmatesmd) | +| Per-run overrides and tuning | [Environment variables](#environment-variables) | + +## FM_HOME + +`FM_HOME` selects the operational home for one firstmate instance. + +| Location | What it contains | Default relationship | +| --- | --- | --- | +| Firstmate repo root | Shared code, including the scripts in this repo's `bin/` | Most scripts also use this as the operational home when `FM_HOME` is unset. | +| Operational home | Private `state/`, `data/`, `config/`, and `projects/` | Selected by `FM_HOME`. | +| Projects directory | Local project clones | Under the operational home; `FM_PROJECTS_OVERRIDE` can select a different directory for tests and specialized harness setup. | + +When `FM_HOME` is unset, most scripts use the repo root as the home. +When it is set, scripts still run from this repo's `bin/`, while `state/`, `data/`, `config/`, and `projects/` come from `$FM_HOME`. + +### Root and directory overrides + +`FM_ROOT_OVERRIDE` overrides the firstmate repo root used by scripts, including the primary checkout watched by the worktree-tangle guard. +When `FM_HOME` is unset, it also behaves as the old whole-root override. + +`bin/fm-send.sh` requires `FM_HOME` to be set before resolving a target. +Unlike most scripts, it does not use the general fallback, because a steer must not silently resolve against the wrong home. +These variables override individual operational directories for tests and specialized harness setup: + +| Variable | Directory selected | +| --- | --- | +| `FM_STATE_OVERRIDE` | Runtime state | +| `FM_DATA_OVERRIDE` | Durable private records | +| `FM_PROJECTS_OVERRIDE` | Local project clones | +| `FM_CONFIG_OVERRIDE` | Local configuration | + +### Relative paths and lifecycle safety + +Before `fm-brief.sh`, `fm-spawn.sh`, or `fm-afk-launch.sh` saves a path or passes it to another process, it handles each applicable `FM_HOME`, `FM_STATE_OVERRIDE`, or `FM_DATA_OVERRIDE` directory as follows: + +- Resolve relative directories against the caller's working directory. +- Preserve accepted absolute spellings unchanged. +- Reject an unresolvable relative directory and name the offending variable. + +`fm-spawn.sh` additionally rejects control bytes in those raw directory inputs before shell or filesystem normalization can change which path the backlog gate checks. +Lifecycle access to a backlog, task record, or pending-close record must resolve within its configured data or state root, and a final-component symlink is refused even when its target remains within that root. -The shared orchestrator behavior lives in [`AGENTS.md`](../AGENTS.md) - edit it like any prompt when the fleet is empty, or dispatch shared-repo edits to a crewmate while tasks are in flight. +Bootstrap applies the same relative `FM_HOME` resolution only when embedding that home in the generated Relay poll shim. +Other transient consumers retain their existing shell-relative behavior. + +### Backend labels and containers + +| Backend | Effect of the operational home | +| --- | --- | +| herdr | `FM_HOME` determines the adapter's workspace label. | +| zellij | `FM_HOME` determines the readable home prefix in visible tab titles, but does not split containers; use `FM_ZELLIJ_SESSION` for a separate session; the full home label also includes a short hash of the resolved `FM_ROOT` path. | +| cmux | `FM_HOME` determines the default config path and readable home prefix in workspace titles; `FM_CONFIG_OVERRIDE` overrides where `config/cmux-socket-password` is read; the full home label also includes a short hash of the resolved `FM_ROOT` path; there is no per-home container split. | ## Operational home layout and state -This section is the single owner of the top-level operational-home layout; producer script headers and their help own exact child-file fields and mutation contracts. -The tracked code root contains the shared instruction, skill, documentation, workflow, and `bin/` surfaces, while each effective `FM_HOME` contains private operational directories. -`data/` holds durable private fleet records such as the project and secondmate registries, captain preferences, optional shared captain preferences, learnings, backlog, briefs, scout reports, and explicitly installed content-addressed extension packages under `data/extensions/packages/`. -`state/` holds runtime records such as task metadata, append-only status events, endpoint signals, watcher and wake-queue coordination, inactive terminal-outcome receipts under `state/terminal-outcomes/`, enabled extension working namespaces under `state/extensions/`, away-mode state, generated Relay artifacts, parent-side remote ledger copies under `state/secondmate-summary-cache/`, one-shot Bearings reconcile requests under `state/reconcile-notify/`, private secondmate config-reread generations with their retry and quarantine state, per-task steering-inbox records under `state/.inbox/` (`bin/fm-task-inbox-lib.sh`), and parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). -`config/` holds local gitignored operating choices, including explicit extension bindings under `config/extensions.d/`, and `projects/` holds the local project clones that Firstmate reads but changes only through the narrow guarded and concrete captain-approved exceptions in `AGENTS.md`. +This section is the single owner of the top-level operational-home layout. +Producer script headers and their help own exact child-file fields and mutation contracts. +The tracked code root contains shared instructions, skills, documentation, workflows, and `bin/`. +Each effective `FM_HOME` contains private operational directories. + +`data/` holds durable private fleet records: + +- Project and secondmate registries. +- Captain preferences and optional shared captain preferences. +- Learnings, backlog, briefs, and scout reports. +- Explicitly installed content-addressed extension packages under `data/extensions/packages/`. + +`state/` holds runtime records: + +- Task metadata, append-only status events, and endpoint signals. +- Watcher and wake-queue coordination, away-mode state, and generated Relay artifacts. +- Inactive terminal-outcome receipts under `state/terminal-outcomes/`. +- Enabled extension working namespaces under `state/extensions/`. +- Parent-side remote ledger copies under `state/secondmate-summary-cache/`. +- One-shot Bearings reconcile requests under `state/reconcile-notify/`. +- Private secondmate config-reread generations with their retry and quarantine state. +- Per-task steering-inbox records under `state/.inbox/` (`bin/fm-task-inbox-lib.sh`). +- Parent-owned secondmate pending-reply records under `state/pending-replies/` (`bin/fm-pending-reply-lib.sh`). + +`config/` holds local gitignored operating choices, including explicit extension bindings under `config/extensions.d/`. + +`projects/` holds local project clones. +Firstmate reads these clones, but changes them only through the narrow guarded and concrete captain-approved exceptions in `AGENTS.md`. Untracked files and directories whose names begin with `scratchpad` are also gitignored, so temporary scratch does not make porcelain-based secondmate sync guards treat a home as dirty. -`bin/fm-spawn.sh` owns the base task-metadata fields it emits, while the runtime-backend section below owns backend-specific fields and selector interpretation. -`bin/fm-contributions.sh` owns durable published-contribution records under each task, observation bounds, equivalent triage-label configuration, and the authenticated contribution check. -The producing PR and Relay helpers own the fields they append, [`bin/fm-classify-lib.sh`](../bin/fm-classify-lib.sh) owns status-event vocabulary, optional emission-time syntax, and legacy unknown-time handling, and `bin/fm-crew-state.sh` owns current-state reconciliation. -The [`bin/fm-fleet-snapshot.sh` header](../bin/fm-fleet-snapshot.sh) owns the snapshot's event-time and age fields, including secondmate parent-event projections. -Wake, watcher, away-mode, and Relay-specific state mechanics remain with their named scripts and reference sections rather than being duplicated into one exhaustive state tree here. +### Format and lifecycle references + +- `bin/fm-spawn.sh` owns the base task-metadata fields it emits, while the runtime-backend section below owns backend-specific fields and selector interpretation. + +- `bin/fm-contributions.sh` owns durable published-contribution records under each task, observation bounds, equivalent triage-label configuration, and the authenticated contribution check. + +- The producing PR and Relay helpers own the fields they append, [`bin/fm-classify-lib.sh`](../bin/fm-classify-lib.sh) owns status-event vocabulary, optional emission-time syntax, and legacy unknown-time handling, and `bin/fm-crew-state.sh` owns current-state reconciliation. + +- The [`bin/fm-fleet-snapshot.sh` header](../bin/fm-fleet-snapshot.sh) owns the snapshot's event-time and age fields, including secondmate parent-event projections. + +- Wake, watcher, away-mode, and Relay-specific state mechanics remain with their named scripts and reference sections rather than being duplicated into one exhaustive state tree here. + +### Session-start references + +- `bin/fm-session-start.sh`'s header is the single owner of session-start ordering, composed commands, digest contents, and the digest's startup mechanism. + +- `bin/fm-startup-network.sh`'s header owns the deferred startup stage that keeps every external-network call and the potentially slow inactive-outcome scan off that digest's blocking path, including its state files and the safety argument for running them later. + +- `docs/sessionstart-nudge.md` owns the native session-open adapter tiers that run or nudge the digest command, and the source routing between them. + +- `AGENTS.md` retains the run-once and read-once operator rules, lock-refusal safety, installation consent, and direct-report recovery boundaries because those facts apply at every session start. + +- Ordinary dead-direct-report recovery is owned by `stuck-crewmate-recovery`, while persistent-secondmate recovery is owned by `secondmate-provisioning`. + +## Orchestrator behavior (AGENTS.md) -`bin/fm-session-start.sh`'s header is the single owner of session-start ordering, composed commands, digest contents, and the digest's startup mechanism. -`bin/fm-startup-network.sh`'s header owns the deferred startup stage that keeps every external-network call and the potentially slow inactive-outcome scan off that digest's blocking path, including its state files and the safety argument for running them later. -`docs/sessionstart-nudge.md` owns the native session-open adapter tiers that run or nudge the digest command, and the source routing between them. -`AGENTS.md` retains the run-once and read-once operator rules, lock-refusal safety, installation consent, and direct-report recovery boundaries because those facts apply at every session start. -Ordinary dead-direct-report recovery is owned by `stuck-crewmate-recovery`, while persistent-secondmate recovery is owned by `secondmate-provisioning`. +The shared orchestrator behavior lives in [`AGENTS.md`](../AGENTS.md). +Edit it like any prompt when the fleet is empty. +While tasks are in flight, dispatch shared-repo edits to a crewmate. ## Calm preference (config/calm) -The Pi Calm extension and the Claude Code Calm mod share the captain's home-local presentation choice in gitignored `config/calm` under the effective Firstmate home, so one `/calm` choice applies on either harness. -Both resolve that home from `FM_HOME`, then `FM_ROOT_OVERRIDE`, then the tracked code root derived from their own path under it, or use `FM_CONFIG_OVERRIDE` as the config directory outright when that test and specialized-setup override is present. -The values they write are `on` and `off`, each followed by one newline; an absent, unreadable, or unrecognized value defaults to off. -`max` is the legacy value written by a removed third presentation level whose behavior is now ordinary Calm, and it is still read as `on`, so a home upgraded from it keeps Calm on rather than dropping to off. -Each `/calm` command persists the new choice before changing live presentation, so a failed write leaves the current choice unchanged rather than claiming persistence; Pi replaces the file atomically, while the Claude Code mod writes it through the plugin API's plain file write. +The Pi Calm extension and the Claude Code Calm mod share the local, gitignored `config/calm` preference under the effective Firstmate home. +One `/calm` choice therefore applies on either harness. +Both resolve the home in this order: `FM_HOME`, `FM_ROOT_OVERRIDE`, then the tracked code root derived from their own path under it. +When `FM_CONFIG_OVERRIDE` is present for tests or specialized setup, it selects the config directory directly. + +### Values and default + +| Value or file state | Result | +| --- | --- | +| `on` | Calm on. | +| `off` | Calm off. | +| Absent, unreadable, or unrecognized | Defaults to off. | + +Both written values end with one newline. +`max` is a legacy value from a removed third presentation level. +Its behavior is now ordinary Calm, so it is still read as `on`. +A home upgraded from `max` keeps Calm on rather than dropping to off. + +### Saving and reloading the preference + +Each `/calm` command saves the new choice before changing live presentation. +A failed write leaves the current choice unchanged and is not reported as a saved preference. +Pi replaces the file atomically; the Claude Code mod uses the plugin API's plain file write. The Pi extension reloads this preference on every Pi `session_start`, including startup, new, resume, fork, and reload reasons. -The Claude Code mod likewise reloads it on every `session.start`, including same-process session replacement, and also loads it lazily before any row that can draw ahead of that event, including during `claude --continue` restoration. + +The Claude Code mod reloads it on every `session.start`, including same-process session replacement. +It also loads the preference lazily before any row that can draw ahead of that event, including during `claude --continue` restoration. This preference is local to each Firstmate home and is not part of secondmate inherited configuration. ## Pi supervision branch -On a Pi primary, an in-process supervision branch handles eligible task-local wake rows and selected heartbeat reviews while keeping main-only rows on the captain-facing path; [docs/pi-supervision-branch.md](pi-supervision-branch.md) owns its conversation lifecycle, row eligibility, mixed-queue dispatch, heartbeat routing, and pre-drain recheck. +On a Pi primary, an in-process supervision branch handles eligible task-local wake rows and selected heartbeat reviews. +Main-only rows stay on the captain-facing path. +[docs/pi-supervision-branch.md](pi-supervision-branch.md) defines its conversation lifecycle, row eligibility, mixed-queue dispatch, heartbeat routing, and pre-drain recheck. Supervision is default-on: once a Pi primary session owns this home's fleet lock, the branch is eligible for every task with no captain grant file required. -A genuinely no-op heartbeat is absorbed in bash and never reaches Pi, and every watcher-failure alarm stays on the captain-facing main path. -A broken branch still falls back to today's wake-to-main path in both postures, and the legacy `state/.afk` daemon flag means nothing on Pi. -While the away-posture record `state/.afk-contract` exists the branch takes every actionable row, no processing turn opens on the parked main, and main's standing authority relocates to the branch through the guarded scripts, each keeping its own gate; [docs/pi-supervision-branch.md](pi-supervision-branch.md#postures) owns that posture. -While attended the branch's role stays bounded exactly as the captain-approved architecture set it: it cannot merge a PR, land local work, freshly spawn, or answer a decision, and every existing captain gate remains unchanged in either posture. + +Bash absorbs a genuinely no-op heartbeat before it reaches Pi. +Every watcher-failure alarm stays on the captain-facing main path. +If the branch breaks, wakes still fall back to main in both postures. +The legacy `state/.afk` daemon flag has no effect on Pi. + +### Attended and away authority + +While the away-posture record `state/.afk-contract` exists: + +- The branch takes every actionable row. +- No processing turn opens on the parked main. +- Main's standing authority moves to the branch through the guarded scripts, each retaining its own gate. + +[docs/pi-supervision-branch.md](pi-supervision-branch.md#postures) defines that posture. + +While attended, the branch cannot merge a PR, land local work, freshly spawn, or answer a decision. +These are the bounds set by the captain-approved architecture. +Every existing captain gate remains unchanged in either posture. Homes on other primary harnesses do not load the Pi branch extension; shared per-task lease behavior is owned by `bin/fm-lease-lib.sh`. + `AGENTS.md`'s `state/` inventory routes the branch's runtime files to their format and lifecycle owners. -While attended, a captain-facing (verdict `captain`) branch outcome persists as one exact, sequence-keyed visible transcript entry and then opens one sequence-keyed processing turn on main, which stays open until main acknowledges that sequence through its `fm_branch_processed` tool; while away, the entry persists but processing waits until the record is archived. + +### Outcome delivery and acknowledgement + +While attended, a captain-facing branch outcome (verdict `captain`) is saved as one exact visible transcript entry keyed by sequence. +It then opens one processing turn on main for that sequence. +The turn stays open until main acknowledges the sequence through its `fm_branch_processed` tool. +While away, the entry is saved, but processing waits until the away-posture record is archived. The branch prompt's "Verdict: routine or captain" section owns the distinction between captain-facing, unsolicited routine, and unchanged-review outcomes. + The generated [Pi supervision protocol](supervision-protocols/pi.md) owns main's event ownership, acknowledgement duty, and conversational treatment for merged outcomes, while the persisted entry itself owns captain visibility. A no-change heartbeat outcome explicitly reported with `task=fleet` and `silent=true` is delivered silently with no rendered note, while every other routine outcome still appends a rendered, sailboat-prefixed note. ## Pi supervision branch model and effort (config/supervision-branch-model, config/supervision-branch-effort) -Supervision is an easier job than the captain's own conversation, so the branch can run on a cheaper model than main. -It is also an easier job than the captain's own conversation needs reasoning for, so the branch can run at a shallower effort than main as well. +The branch can run on a cheaper model than main because supervision is an easier job than the captain's own conversation. +It can also use a lower reasoning effort because supervision needs less reasoning than that conversation. + +### Choose a model and effort + The Pi `/supervision-model` command settles both in one flow: it opens a selector over the models that Pi reports with configured credentials and that this home's stored credentials let the isolated supervision branch resolve, plus a first "Follow main" entry, and then a second picker for the branch's reasoning effort. In Pi's terminal TUI, the model step uses Pi's bounded scrolling list with its input and fuzzy filtering primitives, the same list primitive Pi's `/model` picker scrolls: typing filters the entries, "Follow main" stays the first entry whenever it still matches, and a long catalog scrolls inside the dialog instead of running off the terminal. + The non-TUI RPC, JSON, and print modes have no custom-component surface and keep Pi's generic selector without search, where terminal overflow does not apply. The effort list is a handful of levels and stays on Pi's plain selector dialog. + Both picks change the supervision branch alone and never the captain's own conversation model or effort. -It persists the model pick in gitignored `config/supervision-branch-model` and the effort pick in gitignored `config/supervision-branch-effort`, both under the effective Firstmate home, resolved from `FM_HOME`, then `FM_ROOT_OVERRIDE`, then the tracked code root derived from the extension path, or under `FM_CONFIG_OVERRIDE` when that test and specialized-setup override is present. -Firstmate keeps no model catalog of its own; the list is the intersection of what Pi reports when the picker opens and what a fresh isolated branch runtime can run. + +### Saved settings and available models + +The command saves the model pick in gitignored `config/supervision-branch-model` and the effort pick in gitignored `config/supervision-branch-effort`. +Both live under the effective Firstmate home, resolved in this order: `FM_HOME`, `FM_ROOT_OVERRIDE`, then the tracked code root derived from the extension path. +When `FM_CONFIG_OVERRIDE` is present for tests or specialized setup, it selects the config directory directly. +Firstmate keeps no model catalog of its own. +The list is the intersection of what Pi reports when the picker opens and what a fresh isolated branch runtime can run. + A provider that exists only because an extension registered it inside the captain's session, such as pi-devin-auth's `devin`, is offered and can be pinned or followed like any other; [pi-supervision-branch.md](pi-supervision-branch.md#cost-model-and-the-byte-stable-prefix) owns how that registration reaches the isolated branch runtime. Stored OAuth and API-key credentials retain their native credential type because Firstmate never copies, converts, installs, or overwrites credentials for the branch runtime. -The file holds one `/` line followed by one newline, split at the first `/` so a provider-qualified model id such as `openrouter/anthropic/claude-sonnet-4-5` survives intact. -An absent, unreadable, or unparseable file means no pin, and the branch then follows main's own current model, applied explicitly and live whenever main changes models mid-session. + +### Model file format and default + +The model file holds one `/` line followed by one newline. +Parsing splits at the first `/`, so a provider-qualified model id such as `openrouter/anthropic/claude-sonnet-4-5` survives intact. +An absent, unreadable, or unparseable file means no pin. +The branch then follows main's current model, applied explicitly and live whenever main changes models mid-session. + +### Following a native Codex model + When main uses `codex-native`, following main explicitly selects the same model through ordinary Pi's `openai-codex` provider, so the background branch owns an independent Pi conversation. If that ordinary Pi model is unavailable, the branch refuses to build and returns the notification to main; it never inherits the main native thread or silently selects a different model. + Picking "Follow main" under a `codex-native` main reports that same `openai-codex` model, or that same refusal, because the command and the branch build share one follow rule. A `codex-native` branch pin is refused and excluded from the picker. + +### Applying and changing a model pin + A valid pin wins over main and remains unaffected by main's model changes. Picking "Follow main" removes the file, and the command writes a pin at mode `0600` and replaces it atomically so a failed write leaves the current choice unchanged rather than claiming persistence. -The file's current state decides the branch model on every branch build - the new conversation each main session start opens and the reopen after a model or effort change inside one session - and it overrides Pi's restore of whatever model a reopened branch session recorded, so the choice survives all of them. + +The file decides the branch model on every build: the new conversation opened at each main session start, and a reopen after a model or effort change within a session. +It overrides the model Pi would otherwise restore from the reopened branch session, so the choice survives both cases. That override is what keeps "Follow main" honest: a branch conversation that ran under an earlier pin still records that model, so clearing the file explicitly applies main's model rather than letting the reopened session restore the old one. -For ordinary Pi providers, only when main's own model is unknown, or this home's stored credentials cannot run it in the isolated branch runtime, does an unpinned build fall back to passing no override at all, which is the behavior from before this file existed; the wake is never lost over model choice, and the command says plainly when main's model could not be applied instead of reporting a change that did not take effect. -A pin naming a model Pi cannot hand back, because the model is unknown or has no configured credentials, is never silently downgraded onto main's model: the branch refuses to build and rejects the accepted wake to the watcher's captain-facing main path, exactly as any other unreachable branch does. + +### Unavailable models + +For ordinary Pi providers, an unpinned build passes no model override only when main's model is unknown or this home's stored credentials cannot run it in the isolated branch runtime. +This preserves the behavior from before the file existed. +Model choice never loses the wake. +If main's model could not be applied, the command reports that failure instead of reporting a change that did not take effect. +If a pin names a model Pi cannot return because it is unknown or has no configured credentials, the branch refuses to build. +It sends the accepted wake back to the watcher's captain-facing main path, as any other unreachable branch does. +It never silently falls back to main's model. + Picking also releases the live branch so the next wake reopens this session's own branch conversation under the new model without waiting for a session replacement. -The effort file holds one Pi thinking level followed by one newline, and the two pins are independent: a captain may pin a model, an effort, both, or neither. -The effort step runs after the model step because the effective branch model decides which levels exist: its menu is Pi's own supported-level list, so a model that maps no extended levels simply does not offer them and a non-reasoning model offers only `off`. +### Effort file format and available levels + +The effort file holds one Pi thinking level followed by one newline. +Model and effort pins are independent: a captain may pin either, both, or neither. +The effort step follows the model step because the effective branch model determines which levels exist. +The menu uses Pi's own supported-level list. +Models without extended levels do not offer them; a non-reasoning model offers only `off`. + The picker keeps no effort catalog of its own; when main's model cannot be resolved, it first resolves the model recorded by the most recent branch conversation and uses Pi's supported levels for that effective model. If neither model can be resolved, the picker invents no levels and the command says that the branch's effective effort cannot be determined. -An absent, unreadable, or unrecognized file means no effort pin, and the branch then follows main's own current effort, applied explicitly and live whenever main changes effort mid-session. + +### Applying and changing an effort pin + +An absent, unreadable, or unrecognized file means no effort pin. +The branch then follows main's current effort, applied explicitly and live whenever main changes effort mid-session. A valid pin wins over main and remains unaffected by main's effort changes. + Picking "Follow main" removes the file, and the command writes an effort pin at mode `0600` and replaces it atomically, exactly as it writes a model pin. The effort file's current state decides the branch effort on every branch build, on the same create-and-reopen contract as the model pin and for the same reason: a reopened branch conversation records the effort it last ran under, so only an explicit override keeps "Follow main" honest. + Only when main's own effort cannot be read either does an unpinned build fall back to passing no effort override at all, which is the behavior from before this file existed. -Pi owns the clamp, so a pinned level the branch's model cannot run becomes that model's nearest supported level rather than a refusal; the branch is never refused over effort, the captain's raw pick is kept so it applies again on a model that supports it, and the command reports the level the branch will really run at rather than the raw pin. + +### Unsupported effort levels + +Pi maps an unsupported pinned level to the branch model's nearest supported level. +Effort never causes the branch to refuse a build. +The captain's raw pick is kept so it applies again on a model that supports it, and the command reports the level the branch will actually use. An effort token Pi would not recognize at all is treated as no pin rather than passed to that clamp, which would otherwise collapse a typo into the model's lowest level. +### Cancellation and inheritance + Cancelling the model picker cancels the whole command and changes neither choice. -Cancelling only the effort picker keeps the standing effort choice and still applies the model pick made in the same run, and the command's one closing message reports both choices as they will actually take effect. +Cancelling only the effort picker keeps the standing effort choice and still applies the model pick from the same run. +The command's closing message reports both choices as they will take effect. + Both choices are local to each Firstmate home and are not part of secondmate inherited configuration, the same as the Calm preference; a secondmate home pins its own supervision model and effort with its own `/supervision-model`. ## Supervision host (config/supervision-host) -The optional local, gitignored `config/supervision-host` opts this home into the supervision host, which runs the supervision branch's contract on a headless engine session beside a non-Pi primary; [docs/supervision-host.md](supervision-host.md) owns the design, its current scope, and the verified engines. -Today a Claude, Cursor, OpenCode, omp, Grok, or Codex primary runs it, and only for the away posture: with the file present, that primary's arm owner runs the host in the watcher arm's place, the host handles wakes on the engine while the away-posture record `state/.afk-contract` exists, and `/afk` launches no away daemon on that home, while `/quiet` still does. +The optional local, gitignored `config/supervision-host` enables a supervision host for this home. +The host runs the supervision branch's contract on a headless engine session beside a non-Pi primary. +[docs/supervision-host.md](supervision-host.md) defines its design, current scope, and verified engines. +A Claude, Cursor, OpenCode, omp, Grok, or Codex primary can run the host, only while away. +With the file present, the primary's arm owner runs the host in place of the watcher arm. +The host handles wakes on the engine while `state/.afk-contract` exists. +On that home, `/afk` launches no away daemon; `/quiet` still does. + Absence leaves the home exactly as it is without the host, on every harness; a Pi primary keeps its in-process supervision branch whether or not the file exists. A Grok primary reads the file when its session-start block renders, so a change takes effect at its next session start; every other owner reads it at every arm. + +### Engine selection + The file may be empty, or hold one line ` []`: - empty or `default` selects the primary harness's own engine at that engine's default model (`sonnet` for the Claude engine); @@ -109,8 +320,13 @@ The file may be empty, or hold one line ` []`: Only Claude has a verified engine of its own, so a Cursor, OpenCode, omp, Grok, or Codex home names `claude` in the file. -An engine that is not verified, a primary with no verified engine, or a malformed line leaves the host with no engine: it takes no wake, every wake reaches main as it would without the host, and each away-posture wake carries a line naming the problem. +### Failures and when changes apply + +An unverified engine, a primary without a verified engine, or a malformed line leaves the host without an engine. +It takes no wake, so every wake reaches main as it would without the host. +Each away-posture wake includes a line naming the problem. The file is read at every wake, so a change applies at the next one without a restart. + It is local to each home and not part of secondmate inherited configuration. While the file exists, main's lease-checked commands also take the per-task lease lock, so a claim by the host's engine cannot race a mutation main already started (`bin/fm-lease-lib.sh`). @@ -118,82 +334,184 @@ While the file exists, main's lease-checked commands also take the per-task leas The tracked `.tasks.toml` pins the default `tasks-axi` markdown backend to `data/backlog.md`, with `done_keep = 10` and an archive at `data/done-archive.md`. A home may instead select another tasks-axi adapter such as Beads through its own `.tasks.toml` or `TASKS_AXI_BACKEND`; firstmate still uses only tasks-axi verbs for routine backlog reads and mutations, and the adapter maps `start` and evidence-bearing `done` transitions to its native statuses and evidence fields. + +### Captain holds on Beads + Captain-hold row creation is owned by [`bin/fm-captain-hold.sh`](../bin/fm-captain-hold.sh) `hold`: when no work item exists, it creates an ordinary backlog row (`--kind captain` metadata; Beads native type `task`) and then applies the captain hold. Captain rows have no Beads due semantics, so that create path waives a Beads `due.required` setting rather than passing a synthetic `--due`; `--until` remains the optional hold deferral. + Do not register a Beads `types.custom` `captain` type for this: captain is a hold kind, and the fleet Beads `due.required` policy for ordinary work stays in the federated beads config. -When the automatic transition gate applies, dispatch and completion are not separate operator actions: each moves its work item inside the same run that creates or removes the task's record, so the ordinary successful path cannot leave the backlog and live task set out of sync ([`bin/fm-backlog-transition-lib.sh`](../bin/fm-backlog-transition-lib.sh)). + +### Automatic dispatch and completion + +When the automatic transition gate applies, dispatch and completion each move the work item in the same run that creates or removes its task record. +The ordinary successful path therefore keeps the backlog and live task set in sync ([`bin/fm-backlog-transition-lib.sh`](../bin/fm-backlog-transition-lib.sh)). Under that gate, dispatch accepts only an unheld, unblocked Queued or In flight item in this home; a missing, Done, held, or dependency-blocked item is refused before any endpoint or local copy is created. -[`bin/fm-tasks-axi.sh`](../bin/fm-tasks-axi.sh) refuses `add --start` and its `create --start` alias so neither spelling places a row In flight without dispatch artifacts, because such a row would have no task record, status file, or inbox and would count as live work nobody is doing; `tasks-axi start ` remains a documented direct transition the wrapper passes through. + +[`bin/fm-tasks-axi.sh`](../bin/fm-tasks-axi.sh) refuses `add --start` and its `create --start` alias. +Either would place a row In flight without a task record, status file, or inbox, counting it as live work that nobody is doing. +The wrapper still passes through the documented direct transition `tasks-axi start `. Completion refuses to report success until the item is closed, and session start reconciles this home's own books after an interrupted run. + When a spawn is interrupted after launch delivery began, its exit path re-reads the paired task record and the backlog row under the same per-task lock as the commit, repairs a row the commit believed it had moved, and reports only what was verified or honestly attempted, never intent phrased as outcome ([`bin/fm-spawn.sh`](../bin/fm-spawn.sh); [`tests/fm-backlog-atomicity.test.sh`](../tests/fm-backlog-atomicity.test.sh)). + +### Which backlog receives a transition + Automatic transitions run from the configured data directory's parent, letting that home's effective tasks-axi configuration address its selected adapter while keeping relative scout-report links rooted there. A markdown backlog is additionally addressed by an explicit `--file` at `/backlog.md`, so the change lands in the home that owns the task regardless of the caller's working directory. + Any other configured adapter is addressed by that root alone, because `--file` would override the adapter's own workspace path. -The gate does not apply to persistent secondmates, manual-backend homes, or markdown homes without a backlog file, preserving their existing persistent-agent, manual, or ad-hoc lifecycle behavior while configured non-markdown adapters remain active without that file. + +### Exemptions and refusal conditions + +The gate does not apply to persistent secondmates, manual-backend homes, or markdown homes without a backlog file. +Those retain their existing persistent-agent, manual, or ad-hoc lifecycle behavior. +Configured non-markdown adapters remain active without that file. Migrated-hold resolution on a beads home reads its graph path, binary, and prefix from the root `.tasks.toml` `[beads]` section only, and refuses (rc=2) when the beads backend is selected elsewhere (a `TASKS_AXI_BACKEND` override or user-level config) with no root-level `[beads]` section. + On an automatic-backend home, missing or incompatible `tasks-axi`, an unresolvable configured data directory, or one containing a control byte fails lifecycle work before mutation. An unreadable backend configuration can refuse lifecycle work before the no-backlog exemption applies; repair the configuration named in the diagnostic ([backend resolution contract](../bin/fm-tasks-axi-lib.sh)). + +### Handoffs between homes + Secondmate handoffs bypass that routine-backend choice: `fm-backlog-handoff.sh` keeps only its own fleet-level validation and delegates the item move to `tasks-axi mv`; its [script header](../bin/fm-backlog-handoff.sh) owns route-specific wake outcomes and remote outbox release. It moves in-scope `## Queued` items only and refuses `## In flight` and historical `## Done` records, which stay with their home for pruning or archiving. + Handoff item bodies must use at least two leading spaces, and the helper refuses a selected item with a single-space or tab-indented continuation rather than risk orphaning it. Because bootstrap requires `tasks-axi` on `PATH` on every profile, that delegation works fleet-wide, and the `config/backlog-backend=manual` knob governs firstmate's own hand-editing of its backlog, not this validated helper. + +### Required tools and manual mode + Compatible means the installed build passes the shared version and feature probe owned by [`bin/fm-tasks-axi-lib.sh`](../bin/fm-tasks-axi-lib.sh), including the atomic multi-ID move required by handoff delegation. Bootstrap requires compatible `tasks-axi` on every profile; see "Toolchain" below for missing-tool reporting and silent default-backend behavior. + Set the local, gitignored `config/backlog-backend` file to `manual` to force manual backlog editing and suppress the verbose `BOOTSTRAP_INFO: tasks-axi available` fact, not missing-tool reporting. A `manual` home owns its backlog file outright: the lifecycle transitions above are skipped there, dispatch and completion never fail over the file's contents, and a completed teardown prints the hand edit that is owed instead. + Absent or `tasks-axi` selects the tasks-axi path. On the default markdown adapter, tasks-axi and manual edits produce the same `## In flight`, `## Queued`, and `## Done` sections. +### Using a separate operational home + The tracked `.tasks.toml` paths resolve against the directory tasks-axi runs in, not `FM_HOME`, so a bare `tasks-axi` run from the code root addresses the code root's `data/` whenever the home lives elsewhere. -tasks-axi writes by renaming a temp file over its target, which replaces a symlink with a regular file, so linking the code-root copy into the home forks the queue on the first such write rather than keeping the two in step. -Every routine firstmate backlog command therefore runs through [`bin/fm-tasks-axi.sh`](../bin/fm-tasks-axi.sh), which addresses this home's backlog and archive from any working directory exactly as lifecycle transitions do, and bootstrap reports a code-root `data/backlog.md` or `data/done-archive.md` that is not this home's own file as a `BACKLOG_RECONCILE: code-root ...` line even in a read-only session. +tasks-axi replaces its target by renaming a temporary file over it. +If the target is a symlink, the write replaces it with a regular file. +Linking the code-root copy into the home therefore forks the queue on the first write instead of keeping the copies in sync. + +Run every routine Firstmate backlog command through [`bin/fm-tasks-axi.sh`](../bin/fm-tasks-axi.sh). +Like lifecycle transitions, it addresses this home's backlog and archive from any working directory. +Bootstrap reports a code-root `data/backlog.md` or `data/done-archive.md` that is not this home's own file as a `BACKLOG_RECONCILE: code-root ...` line, even in a read-only session. ## Runtime backend (config/backend / FM_BACKEND) For spawn-capable adapters, the runtime session-provider backend controls where task windows/endpoints are created, captured, sent to, watched, and killed. -`tmux` is the verified reference backend (see [`docs/tmux-backend.md`](tmux-backend.md)); `herdr` has its own required CI lane (see [`docs/herdr-backend.md`](herdr-backend.md)); `zellij`, `orca`, and `cmux` remain experimental spawn backends with no dedicated real-backend CI lane (see [`docs/zellij-backend.md`](zellij-backend.md), [`docs/orca-backend.md`](orca-backend.md), and [`docs/cmux-backend.md`](cmux-backend.md)). + +| Runtime backend | Verification status | Reference | +| --- | --- | --- | +| `tmux` | Verified reference backend | [`docs/tmux-backend.md`](tmux-backend.md) | +| `herdr` | Has its own required CI lane | [`docs/herdr-backend.md`](herdr-backend.md) | +| `zellij` | Experimental; no dedicated real-backend CI lane | [`docs/zellij-backend.md`](zellij-backend.md) | +| `orca` | Experimental; no dedicated real-backend CI lane | [`docs/orca-backend.md`](orca-backend.md) | +| `cmux` | Experimental; no dedicated real-backend CI lane | [`docs/cmux-backend.md`](cmux-backend.md) | + Treehouse remains the worktree provider for tmux, herdr, zellij, and cmux, since herdr, zellij, and cmux are session providers only; Orca provides both the task worktree and terminal endpoint. -New spawns choose the backend in this order: an explicit `--backend` flag that current authority for that exact task alone has authorized (a present captain instruction or the task's own accepted brief; never later-task precedent by analogy), then `FM_BACKEND`, then the first non-empty line of local gitignored `config/backend`, then runtime auto-detection from `$TMUX`, `HERDR_ENV=1`, or cmux runtime signals, then default `tmux`. + +### Backend selection order + +New spawns choose the backend in this order: + +1. An explicit `--backend` flag authorized for that exact task by a present captain instruction or the task's own accepted brief. + A later task cannot inherit that authority by analogy. +2. `FM_BACKEND`. +3. The first non-empty line of local, gitignored `config/backend`. +4. Runtime auto-detection from `$TMUX`, `HERDR_ENV=1`, or cmux runtime signals. +5. Default `tmux`. + If more than one runtime marker is present, detection resolves innermost-first: `$TMUX` is checked before `HERDR_ENV=1`, which is checked before cmux's primary `CMUX_WORKSPACE_ID` marker and its documented fallback signals - tmux or herdr started from inside a cmux terminal is the innermost, currently-executing layer, while cmux itself (a terminal application, not a nestable multiplexer) is always checked last. See [`docs/cmux-backend.md`](cmux-backend.md#runtime-detection) for why cmux can be selected when `CMUX_WORKSPACE_ID` is absent. + Auto-detected Herdr stays silent like tmux, while auto-detected cmux prints a stderr notice naming `config/backend` and `--backend tmux` because cmux remains experimental. Zellij and Orca are never auto-detected; select them by putting the name in a local `config/backend` file, by exporting `FM_BACKEND=`, or by telling the first mate in chat. + +### Accepted backends and secondmate limits + Any value other than `tmux`, `herdr`, `zellij`, `orca`, or `cmux` is rejected until another adapter is implemented and verified. `fm-spawn.sh` accepts `tmux`, `herdr`, `zellij`, `orca`, and `cmux` for ship and scout tasks; `backend=orca` and `backend=cmux` both still refuse `--secondmate` until secondmate launch semantics are designed for each. + `codex-app` is not an accepted runtime backend yet; [`docs/codex-app-backend.md`](codex-app-backend.md) owns the Codex App boundary. + +### Liveness classification + The session-start secondmate liveness sweep and the watcher's secondmate liveness tick use the recovery-grade `fm_backend_agent_state` classifier where verified. The comment above that function in `bin/fm-backend.sh` is the single owner of its detailed state contract and recovery authorization. + The compatibility helper `fm_backend_agent_alive` continues to collapse those detailed results to `alive`, `dead`, or `unknown` for older callers. -A herdr spawn additionally version-gates against the installed `herdr` binary's protocol and requires `jq`, refusing loudly on an incompatible or missing installation. -A zellij spawn additionally version-gates against the installed `zellij` binary's version and requires `jq`, refusing loudly when either is missing or the version is older than 0.44. -A cmux spawn additionally version-gates against the installed `cmux` binary's version, requires `jq`, and requires the control socket to be reachable and accessible (see [`docs/cmux-backend.md`](cmux-backend.md) "Setup" for the one-time socket-access configuration this needs; Automation mode is the recommended socket control mode, with Password mode supported via `config/cmux-socket-password`), refusing loudly and non-retryably on a `cmuxOnly`/unauthenticated socket. + +### Dependency and socket checks + +- A herdr spawn additionally version-gates against the installed `herdr` binary's protocol and requires `jq`, refusing loudly on an incompatible or missing installation. + +- A zellij spawn additionally version-gates against the installed `zellij` binary's version and requires `jq`, refusing loudly when either is missing or the version is older than 0.44. + +- A cmux spawn additionally version-gates against the installed `cmux` binary's version, requires `jq`, and requires the control socket to be reachable and accessible (see [`docs/cmux-backend.md`](cmux-backend.md) "Setup" for the one-time socket-access configuration this needs; Automation mode is the recommended socket control mode, with Password mode supported via `config/cmux-socket-password`), refusing loudly and non-retryably on a `cmuxOnly`/unauthenticated socket. + A backend spawn refusal from a missing dependency, version gate, or unauthenticated socket is terminal for that selected backend; firstmate surfaces it as a blocker instead of silently retrying another backend. + +### Task metadata + Task meta records `backend=` only for a non-default backend; an absent `backend=` means `tmux`, preserving existing default-path meta files. -Every new task records `endpoint_task_id=` as the cleanup binding between the metadata filename and its opaque runtime endpoint. -A herdr task additionally records `herdr_session=`, `herdr_workspace_id=`, `herdr_tab_id=`, and `herdr_pane_id=`. -A zellij task additionally records `zellij_session=`, `zellij_tab_id=`, and `zellij_pane_id=`. -An Orca task additionally records `orca_worktree_id=` and `terminal=`, with `window=fm-` kept as the shared firstmate alias. -A cmux task additionally records `cmux_workspace_id=` and `cmux_surface_id=`. + +- Every new task records `endpoint_task_id=` as the cleanup binding between the metadata filename and its opaque runtime endpoint. + +- A herdr task additionally records `herdr_session=`, `herdr_workspace_id=`, `herdr_tab_id=`, and `herdr_pane_id=`. + +- A zellij task additionally records `zellij_session=`, `zellij_tab_id=`, and `zellij_pane_id=`. +- An Orca task additionally records `orca_worktree_id=` and `terminal=`, with `window=fm-` kept as the shared firstmate alias. + +- A cmux task additionally records `cmux_workspace_id=` and `cmux_surface_id=`. + +### Task selectors + Task selectors for `fm-peek.sh`, `fm-send.sh`, and `fm-crew-state.sh` resolve centrally through `fm_backend_resolve_selector`. A selector containing `:` is passed through as an explicit backend endpoint escape hatch. + Otherwise an exact task id matching `state/.meta` wins before the legacy `fm-` label fallback, so task ids that themselves start with `fm-` route to their own metadata instead of being stripped. A metadata-routed selector returns the recorded backend target (`terminal=` for Orca, otherwise `window=`), and matching explicit targets can still recover the recorded backend when metadata contains the same endpoint. + Only metadata-routed task selectors carry secondmate-marker and Codex-harness context; explicit endpoint escape hatches do not. -These five sentences are the single owner of the task-selector vocabulary; backend guides and other documents point here instead of restating the resolution order. +These rules are the single owner of the task-selector vocabulary. +Backend guides and other documents refer here instead of restating the resolution order. + +### Teardown identity checks + `fm-teardown.sh ` takes a task id directly and validates the complete metadata-only endpoint identity before any runtime dispatch or cleanup mutation. Missing, empty, duplicate, malformed, backend-inconsistent, or task-mismatched endpoint records are preserved and refused. + Legacy tmux metadata remains cleanup-compatible when its exact window name is `fm-`; opaque non-tmux endpoints require their recorded `endpoint_task_id=` binding. + +### Herdr homes and presentation + `FM_HOME` determines Herdr's home label: the primary home uses `firstmate`, and a secondmate home marked by `.fm-secondmate-home` uses `2ndmate-`. [`herdr-backend.md`](herdr-backend.md#watching-and-task-containers) owns launcher-bound workspace placement, the label-only fallback, collision handling, and recovery behavior. + The local `config/herdr-presentation-spaces` file instead opts a home out of, or explicitly in to, Herdr's default-on disposable single-task visual projection; [Presentation spaces](herdr-backend.md#presentation-spaces) owns its accepted values, default, Herdr version floor, migration, behavior, safety limits, recovery contract, and narrow locked session-start cleanup of exact restored idle-shell children. The setting is inherited into secondmate homes under the primary-authoritative contract owned by [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). + For normal herdr operations, `HERDR_SESSION` selects the named session, but destructive test cleanup must not rely on `HERDR_SESSION` alone. Use the explicit guarded cleanup path described in [`docs/herdr-backend.md`](herdr-backend.md) instead of `herdr server stop`. + +### Zellij sessions + For normal zellij operations, `FM_ZELLIJ_SESSION` selects the named session and defaults to `firstmate`. Zellij has no per-home workspace split: primary and secondmate tasks share that one session, and visible tab titles are scoped by the active `FM_HOME` readable label plus a short hash of the resolved `FM_ROOT` path as `fm--`. + Use the guarded cleanup path described in [`docs/zellij-backend.md`](zellij-backend.md) instead of `kill-all-sessions` or `delete-all-sessions`. + +### cmux workspaces + cmux has no session layer at all - one workspace per task, in whatever cmux window is open - and its socket password (when configured) is read from local, gitignored `config/cmux-socket-password` under the effective config directory, never committed. The caller-facing label remains `fm-`, but the actual cmux workspace title is scoped by the active `FM_HOME` readable label plus a short hash of the resolved `FM_ROOT` path as `fm--`. + Test cleanup must use the guarded path in [`docs/cmux-backend.md`](cmux-backend.md#current-operation-and-safety), never enumerate-and-close every workspace. `config/backend` is inherited into secondmate homes under the primary-authoritative contract owned by [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). @@ -201,30 +519,45 @@ Test cleanup must use the guarded path in [`docs/cmux-backend.md`](cmux-backend. The `/afk` sub-supervisor injects escalation digests into firstmate's own pane independently of where new task endpoints are spawned. It currently supports only `tmux` and `herdr` supervisor panes. + Set `FM_SUPERVISOR_BACKEND=tmux|herdr` and `FM_SUPERVISOR_TARGET=` to override both axes explicitly; for herdr the target is `":"`. Without overrides, backend detection uses `$TMUX_PANE` first, then `HERDR_ENV=1` with `HERDR_PANE_ID`, then falls back to `tmux`. + That keeps a tmux pane nested inside herdr on the tmux transport, matching the runtime backend's innermost-first rule. Target detection uses `FM_SUPERVISOR_TARGET`, then `$TMUX_PANE`, then `"${HERDR_SESSION:-default}:${HERDR_PANE_ID}"` under herdr, then the legacy `firstmate:0` tmux fallback with a warning. + Selecting any other supervisor backend, including `zellij`, `orca`, or `cmux`, refuses at daemon startup instead of trying tmux injection primitives against a non-tmux pane. ## Away-mode wedge alarm channels (config/wedge-alarm) When away-mode injection wedges past `FM_MAX_DEFER_SECS`, the sub-supervisor raises a loud, rate-limited alarm. Beyond the durable `state/.subsuper-inject-wedged` marker and the tmux status-line flash, it attempts a configured backend-independent active alert that can reach the captain even when every pane and its backend status-line is unreadable. + +### Channels and overrides + `config/wedge-alarm` (local, gitignored) lists channel directives, one per non-empty, non-comment line; every listed non-`off` channel fires, best-effort. `FM_WEDGE_ALARM_CHANNEL` overrides the file with a single directive. + Directives are `off` (a position-independent kill switch that disables every active alert), `auto`/`default`, `osascript` (macOS Notification Center banner), `herdr` (herdr UI notification), and `command:` (run `` via `sh -c`, summary on `$1` and stdin). An absent file means `auto`, i.e. default-on on macOS: the alarm exists precisely so a wedged away-mode primary is never silent, and it fires at most once per max-defer window after a genuine wedge. + A missing or failing channel logs and falls through to the next, never crashing the daemon. See [`wedge-alarm.md`](wedge-alarm.md) for the current channel reference, [`verification/supervision.md`](verification/supervision.md#wedge-alarm-channels) for active evidence, and [`examples/wedge-alarm`](examples/wedge-alarm) for a copyable config. ## Trace context propagation (config/trace-context / FM_TRACE_CONTEXT) The optional local, gitignored `config/trace-context` presence flag enables default-off native W3C trace-context propagation. + +### Precedence and session boundary + `FM_TRACE_CONTEXT` overrides the file: `1`/`on`/`true`/`yes` enables, any other non-empty value disables, and unset or empty defers to the file. Each locked home session resolves those inputs once, and all spawns from that home use the frozen decision until a new session starts. + +### Secondmate propagation + When launching a Secondmate, the primary copies the presence flag into its home and passes the primary session's frozen decision as a non-empty `FM_TRACE_CONTEXT=on|off` override for the Secondmate's own session start. A Secondmate on a remote route is covered the same way: the primary resolves and records that task's carrier, and the configured host exports it and receives the same enablement snapshot. + The presence flag is session-scoped enablement, so it transfers at launch and is left unchanged by live convergence into a running home. See [`trace-context.md`](trace-context.md) for carrier semantics, supported routes, the manual fleet-restart requirement, the session boundary, and safety limits; `bin/fm-trace-context-lib.sh`'s header owns the exact mechanics, and [`verification/trace-context.md`](verification/trace-context.md) records repeatable evidence. @@ -236,36 +569,50 @@ See [`fleet-ledger.md`](fleet-ledger.md) for the opt-in setup, record contract, The optional local, gitignored `config/turnend-churn-absorb` presence flag opts this home into a default-off third form of positive work evidence in watcher triage. With it present, every referenced task must independently show positive work evidence, and an eligible bare turn-ended task that lacks authoritative proof may satisfy that requirement when its pane content changed since the previous poll. + +### Evidence and time limit + It stays opt-in because the other two proofs read a verdict the harness itself vouches for while this one infers execution from rendered bytes; with the flag absent triage behaves exactly as it did before. `FM_TURNEND_CHURN_ABSORB_SECS` is a positive integer number of seconds, defaults to `900`, and bounds how long one endpoint's turn-ends may ride that evidence before surfacing anyway. + An invalid value fails closed and surfaces the wake. The bound is required rather than cosmetic because churn and pane staleness read the same pane. + The flag is a home-local supervision-noise preference and is not inherited by secondmate homes, which run their own crew mix. [`architecture.md`](architecture.md) owns the triage contract and `bin/fm-watch.sh`'s `signal_turnend_panes_churned` owns the exact evidence and fail-closed boundaries. ## Parked-gate wait deferral (config/wedge-defer-parked-gate) The optional local, gitignored `config/wedge-defer-parked-gate` presence flag opts this home into a default-off second form of wait evidence in the watcher's wedge timer. + +### When a waiting gate defers an alarm + With it present, a provably-working pane about to escalate is also deferred to the `FM_PAUSE_RESURFACE_SECS` recheck cadence when its crew's own current state is a validation gate whose answer is owed to the supervisor and whose decision for that run is still open, and the recheck names the supervisor and the action that clears the lane instead of reporting a suspected wedge. It stays opt-in because the other evidence is the worker's own declaration about its own silence, while this is derived from a pipeline's gate state, so which lanes give up the escalation ladder for it is a home's choice. + With the flag absent the wedge timer spends no fold or current-state read for it, writes no record, and keeps the unchanged escalation schedule, reasons, and `demand-deep-inspection` wording. The flag is a home-local supervision-noise preference and is not inherited by secondmate homes, which supervise their own crew and own that trade separately. + [`architecture.md`](architecture.md) owns the wait-evidence contract and which records may take the ladder away; `bin/fm-watch.sh`'s `wedge_wait_evidence` owns the exact derivation and its fail-closed boundaries. ## Gate defaults (.no-mistakes.yaml) The tracked `.no-mistakes.yaml` sets `test.evidence.store_in_repo: true` and pins `commands.lint` to `bin/fm-lint.sh`, the same owner CI invokes. Storing evidence in the repo publishes each run's test artifacts to the orphan `no-mistakes/evidence` branch and links them from the PR body, instead of keeping them on local disk under the no-mistakes home. + That branch shares no history with code branches, so evidence never enters a pushed feature branch or the default branch; the worktree's `.no-mistakes/` stays local and CI rejects tracked entries under that path. The [`firstmate-coding-guidelines` skill](../.agents/skills/firstmate-coding-guidelines/SKILL.md#no-mistakes-test-configuration) owns why `commands.test` stays absent and targeted validation belongs to the evidence path. + `commands.test` executes code, so no-mistakes honors it only from the default-branch copy of `.no-mistakes.yaml`; a pushed branch cannot change what the gate runs. See [CONTRIBUTING.md](../CONTRIBUTING.md) for the firstmate-specific local test policy and entry points. + Portable shard evidence and coverage rules are in [fm-test-portable-shards.md](fm-test-portable-shards.md); [herdr-backend.md](herdr-backend.md#destructive-lab-safety) owns the real-Herdr lane's isolation boundary, and [runtime-backends.md](verification/runtime-backends.md#herdr) owns active evidence. ## Captain Preferences (data/captain.md / data/captain-shared.md) Domain-local preferences for one captain's fleet live locally in each home's `data/captain.md`; it is gitignored and printed in the session-start context digest after `data/projects.md` and optional `data/secondmates.md`. Before changing it, inspect the current file and curate the matching bullet in place under the internal [`stow` skill's](../.agents/skills/stow/SKILL.md) tiering and archive contract; add a new bullet only for a genuinely new durable preference. + Shared captain preferences that apply across secondmate domains live only in the primary home's optional `data/captain-shared.md`. `secondmate-provisioning` owns its propagation contract, including the required header, read-only secondmate copies, quarantine diagnostics, and the rollout rule that existing homes trim `data/captain.md` by hand after first propagation rather than deleting private content automatically. @@ -273,140 +620,210 @@ Shared captain preferences that apply across secondmate domains live only in the Fleet-local operational facts and gotchas live locally in `data/learnings.md`; it is gitignored and printed after the captain-preference files in the session-start context digest. The file is created lazily on first learning and follows the internal [`stow` skill's](../.agents/skills/stow/SKILL.md) aging-tier and cold-archive contract: inspect the current file first and curate it instead of appending forever. + There is no shared learnings file by captain decision. ## Startup memory budget (config/startup-memory-budget) `config/startup-memory-budget` is the primary-authoritative per-home allowance for the startup prompt-memory surface: `data/captain.md`, `data/captain-shared.md`, and `data/learnings.md` together. The locked mutable bootstrap path materializes its visible default of `7500` estimated tokens in a primary home when the file is absent. + +### Set and validate the budget + To select another allowance, replace the primary home's file with one valid positive value in the exact format below; the next locked bootstrap convergence or `bin/fm-config-push.sh` propagates it to registered secondmates. A secondmate does not create an independent default and instead receives the primary value through the inherited-local-material contract in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md). + The file must be one positive base-10 integer followed by exactly one newline in a regular, single-linked file beneath a non-symlinked `config/` directory. Malformed, multi-line, symlinked, hardlinked, special, or otherwise unsafe values are rejected rather than treated as a default. + +### Accounting and curation + Use `bin/fm-startup-memory-budget.sh read` to validate and print the effective value, or `bin/fm-startup-memory-budget.sh report` to account for the three files. The stable local estimate is `ceil(UTF-8 bytes / 3)` per file, a conservative portable approximation rather than a provider-exact tokenizer. + An inherited `data/captain-shared.md` counts in a secondmate's total but remains primary-owned and read-only there. The internal [`/stow` skill](../.agents/skills/stow/SKILL.md) owns curation and its automatic secondmate cascade, which accounts every home against this same per-home allowance separately rather than against a fleet total. + The helper's header owns exact parsing, publication, and report output mechanics. ## Stow pass horizon (config/stow-pass-horizon) `config/stow-pass-horizon` is an optional local, gitignored presence flag that opts this home in to the pass-count decay horizon in the internal [`/stow` skill](../.agents/skills/stow/SKILL.md). Without it a `/stow` pass decays memory entries on their wall-clock horizons alone - 30 days for `aging`, 7 days for `perishable` - which is the default and unchanged behavior. + With it, an entry is also stale after 10 passes (`aging`) or 3 passes (`perishable`) that evaluated it without reinforcing it, whichever horizon it reaches first. Opt in for a home that stows often enough that entries never sit unreinforced for a wall-clock horizon, so memory only grows against the startup-memory budget above; a home that stows rarely already exceeds its date horizon on a single pass and gains nothing. + The flag is per home and is not inherited by secondmate homes, because stow cadence is a property of the home doing the stowing. Only the file's presence is read, so its contents are ignored; remove it to return to the default contract on the next pass. + The skill text owns the marker spelling, the tick order, and the reinforcement rule. ## Secondmate routes (data/secondmates.md) Persistent secondmate routes live locally in `data/secondmates.md`. The concise single-line route contract is owned by the [`secondmate-provisioning` skill](../.agents/skills/secondmate-provisioning/SKILL.md#routing-table), including the parser-compatible fields, one-sentence summary requirement, `home:` pointer to the seeded charter, and limit on extra registry prose. + +### Remote routes and validation + A remote route adds `host:` and `root:` before the existing fields and places the whole secondmate home on that SSH host; it does not make ordinary workers remotely placeable. [`remote-secondmates.md`](remote-secondmates.md) owns current remote setup, operation, and safety behavior. + Use `fm-home-seed.sh validate` to check the complete operational registry contract documented by the command itself. The main first mate routes by reading those scopes with judgment; the project list is provisioning data, not exclusive ownership. + +### Provision a local home + Use `fm-home-seed.sh - {...|--no-projects}` to lease a fresh local firstmate worktree for the secondmate home. For remote provisioning, including supplied project origins, follow [Remote second mates](remote-secondmates.md#provision-a-route). + Use the deliberate `--no-projects` signal only for a firstmate-repo domain that needs no separate project clones. It cannot be combined with a project list, and omitting both still fails loudly. + A project-less seed requires no existing project clones or `data/projects.md` entries in the home, so it refuses a populated-home conversion without changing that home. A preexisting project-bearing charter is also refused until it is re-scaffolded with `--no-projects` or removed. + The lease is held under the secondmate id until explicit retirement or seed rollback returns it, so normal restarts do not free or recycle the home. Teardown of a leased home fails closed if `treehouse return` cannot release the lease; plain-clone homes with no treehouse pool slot are removed directly. + +### Project modes and backlog handoff + Secondmate routes cover `no-mistakes` and `direct-PR` projects; `local-only` projects remain main-firstmate work. For `no-mistakes` projects, seeding initializes only projects newly cloned into a secondmate home and refuses to mutate a preexisting clone that is not already initialized. + After creating a secondmate, move existing main-backlog queued items that you have judged in-scope with `fm-backlog-handoff.sh ...`; it refuses In flight, Done, or non-secondmate homes, and its [script header](../bin/fm-backlog-handoff.sh) owns route-specific wake outcomes and retries. Set `FM_SECONDMATE_CHARTER` to seed from inline charter text when no filled charter brief exists; set `FM_SECONDMATE_SCOPE` when the routing scope should differ from the charter text. + The seeded home's `data/charter.md` owns the standard secondmate lifecycle and escalation contract; the route file points to it through the existing `home:` field instead of adding another pointer. + +### Identity markers and upgrades + Each seed writes an `.fm-secondmate-home` identity marker at the home root, alongside a durable `.fm-secondmate-parent` record of the home's route to its parent (see "Provision a route" in [`docs/remote-secondmates.md`](remote-secondmates.md)). The tracked root `.gitignore` ignores both markers, so validation can read them without making a freshly seeded home appear dirty to porcelain-based safety checks. + This does not relax protection for any other untracked file. An existing linked-worktree home that predates this rule advances through its marker-only state during its next bootstrap or spawn local sync, after which Git ignores the marker normally. -A local standalone-clone home cannot receive a primary-local commit through that no-fetch sync, so it receives the rule through `/updatefirstmate`'s origin refresh instead. - -## FM_HOME -`FM_HOME` selects the operational home for one firstmate instance. -When it is unset, most scripts use the repo root as the home; when it is set, scripts still run from this repo's `bin/`, but `state/`, `data/`, `config/`, and `projects/` come from `$FM_HOME`. -`FM_ROOT_OVERRIDE` overrides the firstmate repo root used by scripts, including the primary checkout watched by the worktree-tangle guard. -When `FM_HOME` is unset, it also behaves as the old whole-root override. -`bin/fm-send.sh` is intentionally stricter than that general fallback: it requires `FM_HOME` to be set before resolving a target, so operator steers cannot silently resolve against the wrong home. -`FM_STATE_OVERRIDE`, `FM_DATA_OVERRIDE`, `FM_PROJECTS_OVERRIDE`, and `FM_CONFIG_OVERRIDE` override individual operational directories for tests and specialized harness setup. -Before `fm-brief.sh`, `fm-spawn.sh`, or `fm-afk-launch.sh` persists a path or passes it to another process, it resolves each applicable relative `FM_HOME`, `FM_STATE_OVERRIDE`, or `FM_DATA_OVERRIDE` directory against the caller's working directory, preserves accepted absolute spellings unchanged, and rejects an unresolvable relative directory with the offending variable named. -`fm-spawn.sh` additionally rejects control bytes in those raw directory inputs before shell or filesystem normalization can change which path the backlog gate checks. -Lifecycle access to a backlog, task record, or pending-close record must resolve within its configured data or state root, and a final-component symlink is refused even when its target remains within that root. -Bootstrap applies the same relative `FM_HOME` resolution only when embedding that home in the generated Relay poll shim; other transient consumers retain their existing shell-relative behavior. -For the herdr backend, `FM_HOME` also determines the workspace label used by the adapter. -For the zellij backend, `FM_HOME` does not split containers, but it determines the readable home prefix embedded in visible tab titles; use `FM_ZELLIJ_SESSION` when a separate zellij session is needed. -The full zellij home label also includes a short hash of the resolved `FM_ROOT` path. -For the cmux backend, `FM_CONFIG_OVERRIDE` overrides where `config/cmux-socket-password` is read from, while `FM_HOME` determines the default config path and readable home prefix embedded in workspace titles. -The full cmux home label also includes a short hash of the resolved `FM_ROOT` path, and there is no per-home container split. +A local standalone-clone home cannot receive a primary-local commit through that no-fetch sync, so it receives the rule through `/updatefirstmate`'s origin refresh instead. ## Harness support claude, codex, opencode, pi, pi-signed, grok, kimi, cursor, and omp are empirically verified for crewmate and secondmate launches; gemini is verified for crewmate and scout launches only, and [README requirements](../README.md#requirements) own the set supported for the primary session. + +### Harness restrictions and credentials + `fm-spawn.sh` refuses kimi on cmux and Orca at preflight, because answering Kimi's folder-trust dialog needs a verified viewport-only capture those backends lack; [its adapter reference](../.agents/skills/harness-adapters/references/harness/kimi.md#readiness-gated-start) owns the trust-dialog handling. A cursor secondmate or primary runs the tracked project-scope `.cursor/hooks.json` in its own home and must be launched with `--trust`, or no project hook loads; [`docs/supervision-protocols/cursor.md`](supervision-protocols/cursor.md) owns its supervision protocol. + Cursor typed-submit confirmation is verified on tmux and Herdr only. On Zellij, cmux, and Orca a typed-plane Cursor send (a harness-native invocation or an explicit backend target; ordinary text steers ride the durable inbox and exit 0 at enqueue) lands, but `fm-send` reports delivery unconfirmed and exits non-zero because their shared submit core does not consult the busy footer; [runtime backend verification](verification/runtime-backends.md#cursor-agent-cli) owns the evidence and transcript-state boundary. + muse is verified for crewmate and scout launches ONLY, and `fm-spawn.sh` refuses it for a secondmate, because muse ships no usable hook surface for a primary session's turn-end supervision; [`docs/verification/muse.md`](verification/muse.md) owns that evidence. muse also needs a worker-reachable credential before spawning, and the portable fleet path is the `/muse/auth.json` credential stored by `muse login`, because a caller-only `META_API_KEY` does not cross a long-lived backend daemon. + gemini is likewise refused for secondmates because it has no primary supervision protocol; [its adapter reference](../.agents/skills/harness-adapters/references/harness/gemini.md) owns the credential precondition, canonical-launch wiring, and raw-launch limitations. rovo is likewise verified for crewmate and scout launches ONLY, refused for a secondmate for the same reason - no turn-end hook and no primary supervision protocol; [`docs/verification/rovo.md`](verification/rovo.md) owns that evidence, including the OAuth token's silent background refresh from a stored refresh token and both tmux and herdr pane liveness (herdr placement is verified live, with a Herdr-side agent-detection gap left open for recovery classification). + agy is likewise verified for crewmate and scout launches ONLY, refused for a secondmate for the same reason - no hook surface and no primary supervision protocol; [`docs/verification/agy.md`](verification/agy.md) owns that evidence, including the spawn-time worktree trust pre-registration through `bin/fm-agy-trust.sh` and Herdr's native agy pane recognition. devin is verified for crewmate and scout launches only; a secondmate is refused because Devin has no verified primary supervision protocol. + Its private worker config disables Claude Code imports (including the captain's hooks) and Devin commit attribution without editing user or project config; [`fm-devin-config.sh`](../bin/fm-devin-config.sh) owns these enforced settings and [Devin verification](verification/devin.md) owns the live evidence and observed model availability. + +### Verification and primary supervision + New harnesses get verified through a supervised trial task before joining the set. The verified adapter evidence - each harness's busy-state source, interrupt and exit behavior, skill-invocation syntax, and per-harness quirks - lives in the skill tree rooted at [`.agents/skills/harness-adapters/SKILL.md`](../.agents/skills/harness-adapters/SKILL.md). + The executable interrupt and exit mechanics live in [`bin/fm-control-lib.sh`](../bin/fm-control-lib.sh), and [`docs/agent-control.md`](agent-control.md) owns their lifecycle-control architecture. Launch mechanics, including the verified command templates, live in [`bin/fm-spawn.sh`](../bin/fm-spawn.sh). + Pi-family launches adapt the regular-TUI safeguard to the installed CLI's capabilities; [`fm-spawn.sh --help`](../bin/fm-spawn.sh) owns the exact version-safe launch mechanics. Enabled primary-session turn-end guard integrations are tracked as repo-level hook files and documented in [`docs/turnend-guard.md`](turnend-guard.md). + Kimi remains outside the primary turn-end guard integrations; [`docs/turnend-guard.md`](turnend-guard.md#compatibility-limits) owns its separate captain-approved crew wake hook. Primary-session watcher wake protocols are rendered at session start by [`bin/fm-supervision-instructions.sh`](../bin/fm-supervision-instructions.sh) from [`docs/supervision-protocols/`](supervision-protocols/). + Claude's Stop `asyncRewake` hook owns tokenless re-arm cycles, Cursor's stop hook parks on the watcher, Grok uses background-notify cycles, Codex uses bounded foreground checkpoints, Pi and pi-signed use the same two tracked primary extensions, omp uses its own two tracked `.omp/extensions/` files with a blocking `session_stop` turn-end hook, and OpenCode uses its TUI plugin. + +### Choose the worker harness + `config/crew-harness` is a local, gitignored file containing one adapter name for crewmate and scout launches. When pi-signed is selected, Firstmate preserves `FM_PI_HARNESS=pi-signed` and refuses the launch if the selected executable is unavailable rather than falling back to pi; [`fm-spawn.sh --help`](../bin/fm-spawn.sh) owns executable resolution and launch mechanics. + Plain Pi launches set `FM_PI_HARNESS=pi`, so a signed primary's environment cannot relabel a plain Pi worker. When it is absent or contains `default`, crewmates mirror the firstmate's own harness. + +### Choose the secondmate harness + `config/secondmate-harness` is a separate local, gitignored file containing the adapter the primary uses to launch secondmate agents, optionally followed by model and effort tokens on the same line. The first non-empty, non-comment line is parsed as ` [] []`. + A bare `` preserves the previous behavior: harness only, with no model or effort launch flag. When the harness token is absent or `default`, secondmate launch falls back through `config/crew-harness` and then the primary's own harness, and no model or effort is read from that file. + `fm-harness.sh secondmate-model` and `fm-harness.sh secondmate-effort` expose only the optional tokens from `config/secondmate-harness`; `config/crew-harness` remains a bare adapter-name file. Changing this pin affects the next secondmate spawn or control-plane relaunch; the relaunch profile rules are owned by [`docs/agent-control.md`](agent-control.md#transactional-relaunch). + +### Per-launch overrides and inherited defaults + An explicit harness argument to `fm-spawn.sh` still overrides either config file for that spawn only. An explicit `--model` or `--effort` overrides the matching token from `config/secondmate-harness`; for a local route, an explicit harness or raw launch command starts with clean model and effort defaults unless those flags are also passed. + Remote secondmate routes accept verified harness adapters only and reject raw launch commands. When `config/crew-dispatch.json` exists, crewmate and scout spawns require an explicit resolved harness instead of automatically falling back to `config/crew-harness`. + The inherited-local-material contract is owned by [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md); its harness-relevant consequence is that a secondmate's own crewmates use the primary's dispatch profiles and static harness value. Those inherited values are defaults and rules only; `fm-spawn` still permits a consciously chosen explicit runtime outside the config. + `config/secondmate-harness` is not inherited because secondmates do not launch secondmates. + +### Installed hooks and launch details + For grok, `fm-spawn.sh` installs one firstmate-owned global turn-end hook under `$GROK_HOME/hooks/`, or `~/.grok/hooks/` when `GROK_HOME` is unset, and drops a per-task `.fm-grok-turnend` pointer in the worktree, with teardown removing the task token and pointer. For Kimi crews, `fm-spawn.sh` runs `fm-kimi-turnend-hook.sh install`, drops a per-task `.fm-kimi-turnend` pointer in the worktree, and records the matching private registry token for teardown. + Kimi continues to use the captain's normal Kimi home, including the existing config, skills, and memory; Firstmate does not create an isolated Kimi home. The Kimi installer requires an existing regular non-symlink `~/.kimi-code/config.toml`, `python3` with `tomllib`, and `jq`; it validates but never serializes the captain's TOML and refuses before writing when the config is missing, malformed, or surprising or when either tool requirement is unavailable. + Its `remove` action excises only the marker-delimited Firstmate region and removes Firstmate's hook files. For Pi and pi-signed secondmate launches, `fm-spawn.sh` starts the selected executable with `-e` pointed at the secondmate home's own tracked `.pi/extensions/fm-primary-pi-watch.ts` and `.pi/extensions/fm-primary-turnend-guard.ts`, both already present from the secondmate home's git worktree. + For omp secondmate launches, `fm-spawn.sh` passes no `-e` at all: omp auto-discovers the home's tracked `.omp/extensions/` with no trust gate, and naming a discovered file with `-e` as well loads it twice; every omp launch instead carries the tracked `.omp/fm-worker-overlay.yml` posture overlay through `--config`, which [`fm-spawn.sh --help`](../bin/fm-spawn.sh) owns. ## Claude permission mode (config/claude-permission-mode) -The optional local, gitignored `config/claude-permission-mode` holds one token selecting the permission flag every Claude worker launch carries: crewmates, scouts, Claude secondmates, and control-plane relaunches alike. +The optional local, gitignored `config/claude-permission-mode` selects the permission flag for every Claude worker launch: crewmates, scouts, Claude secondmates, and control-plane relaunches. + +### Accepted values and refusals + The token is the file's whitespace-trimmed content. -`bypass` keeps today's launch, `claude --dangerously-skip-permissions`, and is also the default when the file is absent, so an unconfigured home launches byte-for-byte as before. -`auto` replaces that flag with `--permission-mode auto`, Claude Code's classifier-reviewed permission mode, for a captain who refuses to run workers in bypass mode; every other part of the Claude launch, including its environment prefix, inline settings, model, and effort flags, is unchanged. -Any other value, or an unreadable file, refuses every spawn from that home, whichever harness it would launch, before any endpoint, worktree, or task record exists, and names the accepted values; Firstmate never falls back to a permission posture the captain did not choose. + +| Token | Launch permission flag | +| --- | --- | +| `bypass` | `claude --dangerously-skip-permissions` | +| `auto` | `--permission-mode auto` | + +An absent file defaults to bypass, so an unconfigured home launches byte-for-byte as before. +Auto is Claude Code's classifier-reviewed permission mode, for a captain who refuses to run workers in bypass mode. +Only the permission flag changes. +The environment prefix, inline settings, model, effort flags, and every other part of the Claude launch stay unchanged. + +Any other value or an unreadable file refuses every spawn from that home, whichever harness it would launch. +This happens before any endpoint, worktree, or task record exists. +The diagnostic names the accepted values; Firstmate never falls back to a permission posture the captain did not choose. + +### When changes apply and inheritance + `bin/fm-spawn.sh` reads the file on every spawn and relaunch, so a change takes effect at the next launch without a restart. The file is a captain-wide safety preference, so it is inherited into secondmate homes under the [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md) inherited-local-material contract; a secondmate's own Claude crewmates then launch on the same posture. + The [Claude adapter reference](../.agents/skills/harness-adapters/references/harness/claude.md) records the verified shape of both launches and which once-per-machine dialog each one can meet. ## Worker account pin (config/claude-account, config/pi-account) A home that mixes accounts for one runner, such as a work login and a personal one, can pin the account its own Claude and Pi workers launch on. The pin is opt-in: with neither file, every launch is unchanged, and Claude workers keep receiving firstmate's own `CLAUDE_CONFIG_DIR` when it is set. + Both files are local and gitignored. | Runner | File | Variable the launch receives | `ordinary` means | @@ -414,57 +831,84 @@ Both files are local and gitignored. | `claude` | `config/claude-account` | `CLAUDE_CONFIG_DIR` | the variable unset, so Claude uses its default login | | `pi`, `pi-signed` | `config/pi-account` | `PI_CODING_AGENT_DIR` | `~/.pi/agent` | +### File format and provider selection + `config/claude-account` holds one line: `ordinary`, or the absolute path of an existing Claude config directory. `config/pi-account` holds that same root on line 1 and, on line 2, the providers this home may spend, separated by spaces, for example `openai-codex anthropic`. + A final newline is optional; any other line, a relative path, or a control character such as a CR refuses. For Claude, `ordinary` unsets `CLAUDE_CONFIG_DIR` rather than pointing it at `~/.claude`, because Claude reads `$CLAUDE_CONFIG_DIR/.claude.json` and keys its macOS Keychain entry to any directory that is set ([authentication, "Credential management"](https://code.claude.com/docs/en/authentication#credential-management)). + A Pi root can hold several provider logins at once, so the root alone does not say which account a launch spends. A pinned Pi launch therefore needs `--model /` naming a declared provider, and Firstmate also passes `--provider ` so Pi cannot resolve the model under another signed-in provider. + An unqualified model, an undeclared provider, or a raw Pi launch command, which cannot receive that flag, refuses; Firstmate never guesses a provider. +### Launch scope and sign-in checks + When a file is present, every launch of that runner from this home uses it: ships, scouts, local secondmate agents, raw Claude launch commands, and relaunches. -A raw Claude launch command whose leading assignments set `CLAUDE_CONFIG_DIR` or one of the credentials a pinned launch unsets, such as `ANTHROPIC_API_KEY`, would override the pin, so it refuses and names the variable; remove the assignment from the raw command, or change or remove `config/claude-account`. +A raw Claude launch command refuses if its leading assignments set `CLAUDE_CONFIG_DIR` or a credential that a pinned launch unsets, such as `ANTHROPIC_API_KEY`. +The assignment would override the pin. +The refusal names the variable; remove that assignment from the raw command, or change or remove `config/claude-account`. + Before any worker endpoint, local copy, or task record exists, and before a relaunch stops the running worker, Firstmate asks the runner itself whether the pinned account is signed in: `claude auth status` for Claude, and `pi auth check --provider --json --no-refresh` for Pi, falling back to `pi --list-models ` for a provider an extension registers. The check runs with only `HOME`, `PATH`, `TMPDIR`, `USER`, `LOGNAME`, and the pinned root in its environment, so a credential variable in firstmate's own environment cannot answer for an empty root. + A pinned Claude launch also unsets the environment credentials Claude ranks above a stored login, such as `ANTHROPIC_API_KEY`, `CLAUDE_CODE_OAUTH_TOKEN`, and the Bedrock and Vertex switches ([authentication precedence](https://code.claude.com/docs/en/authentication#authentication-precedence)). Pi ranks a root's stored logins above environment variables, so a pinned Pi launch unsets nothing. + A home that authenticates Claude through environment credentials on purpose should leave the pin absent. +### Failures, reporting, and inheritance + A malformed file, a root that is not a readable directory, or a signed-out account refuses the launch and names the file to fix; Firstmate never falls back to the ambient account and never changes a global login or copies a credential. The spawn prints the pin as `account=` (plus `account_provider=` for Pi) and records the same fields in the task record, so the session-start digest shows which account each worker launched on. + Pins are not inherited into secondmate homes: a local secondmate agent launches on the launching home's pin, while the secondmate's own workers read the secondmate home's files. A remote secondmate is launched on its host from its own home's configuration, so create the file in that remote home. + [`bin/fm-worker-account-lib.sh`](../bin/fm-worker-account-lib.sh) owns parsing, the sign-in check, and the full list of credentials a Claude launch unsets; [runtime backend verification](verification/runtime-backends.md#worker-account-pin-sign-in-check) records the check against the real runners. ## Lavish server address (config/lavish-axi-host) The optional local, gitignored `config/lavish-axi-host` contains one non-empty address without whitespace for the per-machine Lavish server. `fm-spawn.sh` exports that address into every new worker and relaunch for opening boards, and the file is inherited into secondmate homes through the primary-authoritative configuration contract. + Once a board exists, the process-event adapter derives the polling address from that board's own saved Lavish session instead; its header owns the lookup contract. When the file is absent, worker launches do not add a board address and retain the existing ambient-environment behavior. + Malformed or unreadable values refuse the launch before the worker starts. The address selects the existing shared server; it does not authorize starting or stopping the server, and the Lavish startup crash remains a vendor-tool concern. ## Home brief include (config/brief-include.md) -The optional local, gitignored `config/brief-include.md` carries standing worker instructions that one captain wants on every ship and scout brief, so private brief content needs no edit to a tracked file. +The optional local, gitignored `config/brief-include.md` adds standing worker instructions to every ship and scout brief. +This keeps private brief content out of tracked files. When the file exists, `bin/fm-brief.sh` appends its text verbatim as the scaffold's last section, `# Home brief additions`, which defers to every other section of the brief, including the ship contract a later scout promotion appends below it. + An absent or blank file changes nothing, while a present path that is not a readable regular file, or text carrying its own `Delivery contract: mode=` line, stops the scaffold before anything is written. The text is static and never executed or expanded; secondmate charters never take it, and the file is local to each home rather than part of secondmate inherited configuration. + `bin/fm-brief.sh`'s header owns the placement rule and its safety argument. ## Worker launch environment (config/launch-env-allowlist) The optional local, gitignored `config/launch-env-allowlist` limits the ambient environment passed to newly launched workers, scouts, and secondmates, including relaunches. With no file, ambient inheritance remains unfiltered: selected harness markers are cleared, while the provider, long-lived terminal daemon, and shell initialization determine which other variables reach the worker. + Do not assume every worker inherits the invoking Firstmate process's current environment. The file is inherited into secondmate homes through the [primary-authoritative configuration contract](../.agents/skills/secondmate-provisioning/SKILL.md). + Changes apply to subsequent launches; existing processes keep their environment. +### Allowlist format + Create the file with one environment variable **name** per line, never credential values, assignments, wildcards, or shell commands. Blank lines and lines beginning with `#` are allowed. + Invalid names, an unreadable or nonregular file, or a path inspection error (including an inaccessible configuration directory) stop the launch. An empty file enables filtering with only Firstmate's operational floor. + For example, a provider using `OPENAI_API_KEY` and Git using an SSH agent could use: ```text @@ -474,13 +918,19 @@ OPENAI_API_KEY SSH_AUTH_SOCK ``` +### Variables retained and where values come from + Firstmate retains basic home, executable search, terminal, locale, temporary-directory, and backend routing variables, plus its explicit launch assignments, its ship and scout task marker, the compact-adviser kill switch described below, and enabled task trace. [`fm-spawn.sh --help`](../bin/fm-spawn.sh) owns the exact retained names and parsing mechanics. + Other ambient names must be listed explicitly, including custom credential-store locations, proxy settings, and certificate overrides when required by the selected tools. The command shell and worker may still create their own variables. + Allowed values come from the destination pane at execution time; they are neither copied from the invoking Firstmate process nor written into the launch command. Listing a name does not provision it in a daemon's environment or transfer credentials to another machine. +### Authentication requirements + Choose the minimum additions for the authentication method actually in use: | Provider or Git transport | Additional names needed | @@ -493,16 +943,24 @@ Choose the minimum additions for the authentication method actually in use: | Git over SSH with a key file | No credential variable when normal SSH configuration selects the key; file permissions and any passphrase handling still apply. | | Git over HTTPS with a credential helper | Whatever the configured helper requires; a GitHub CLI helper using an environment token needs its selected `GH_TOKEN` or `GITHUB_TOKEN`. | +### Validation and security limits + Verify the selected provider login and Git transport after opting in; Firstmate does not infer credentials from model names or install a secret manager. Raw launch commands run under noninteractive POSIX `sh` with this option and must use compatible syntax. + The filter runs at the worker command boundary, after the terminal daemon and pane shell have started; it does not scrub either of those processes. This is not a sandbox: it cannot revoke same-user access to credential files, prevent tools or later shells from loading credentials again, or isolate processes from the same user's other processes. + Regression coverage executes emitted launch commands with synthetic nonsecret values in [`tests/fm-spawn-dispatch-profile.test.sh`](../tests/fm-spawn-dispatch-profile.test.sh). +### Compact adviser setting + Every crewmate, scout, and secondmate Firstmate launches starts with `COMPACT_ADVISER_DISABLE=1` in its environment, on a fresh spawn and on a relaunch alike, so an unattended session never activates the compact adviser. This guarantee also covers raw launch commands, remote secondmates, and launches filtered by `config/launch-env-allowlist`; it does not depend on the destination environment already containing the variable. + Firstmate provides no configuration or flag to change this value. This applies only to agents Firstmate launches; the captain's own primary Firstmate session is never given the variable. + [`fm-spawn.sh --help`](../bin/fm-spawn.sh) owns the delivery mechanics, with focused regression coverage in [`tests/fm-spawn-compact-adviser-disable.test.sh`](../tests/fm-spawn-compact-adviser-disable.test.sh) and [`tests/fm-spawn-compact-adviser-disable-remote.test.sh`](../tests/fm-spawn-compact-adviser-disable-remote.test.sh). Every claude launch's inline `--settings` JSON also carries `"attribution":{"commit":"","pr":"","sessionUrl":false}`, so a spawned worker never writes a Co-Authored-By trailer, Claude-Session link, or generated-with line into a commit or PR body regardless of which settings scopes end up loaded. @@ -510,10 +968,17 @@ Every claude launch's inline `--settings` JSON also carries `"attribution":{"com ## Crew dispatch profiles (config/crew-dispatch.json) `config/crew-dispatch.json` is an optional local, gitignored file containing natural-language rules that firstmate reads before dispatching a crewmate or scout. -The shell scripts do not match those rules; firstmate chooses the best matching rule with judgment, resolves its profile object or array under the operating contract in `AGENTS.md` section 4 and `quota-array-dispatch`, and passes only concrete `--harness`, `--model`, and `--effort` flags to `fm-spawn.sh`. -When the file exists, `fm-spawn.sh` enforces that contract by refusing crewmate and scout spawns that lack an explicit harness (`--harness`, a positional adapter, or a raw launch command). -Batch spawns satisfy the same requirement with a shared `--harness`. -Secondmate spawns are exempt and still resolve through `config/secondmate-harness` and its optional model and effort tokens. +Firstmate chooses the best matching rule with judgment; shell scripts do not match the natural-language rules. +Firstmate resolves the rule's profile object or array under `AGENTS.md` section 4 and `quota-array-dispatch`, then passes only concrete `--harness`, `--model`, and `--effort` flags to `fm-spawn.sh`. + +**Spawn requirements** + +- When the file exists, `fm-spawn.sh` enforces that contract by refusing crewmate and scout spawns that lack an explicit harness (`--harness`, a positional adapter, or a raw launch command). +- Batch spawns satisfy the same requirement with a shared `--harness`. +- Secondmate spawns are exempt and still resolve through `config/secondmate-harness` and its optional model and effort tokens. + +**Contract owners** + This section is the single owner of the canonical schema and its per-field semantics. `AGENTS.md` section 4 owns the always-loaded dispatch intake boundary, and `quota-array-dispatch` owns the completion-aware profile-array selection procedure. @@ -537,129 +1002,273 @@ This section is the single owner of the canonical schema and its per-field seman } ``` -Per rule, `when` and `use` are required; the top-level `rules` array itself may be absent or empty for a default-only configuration. -Both `use` and the optional top-level `default` accept either one profile object or a non-empty array of profile objects. -The single-object form stays fully backward-compatible, and every profile needs `harness`. -Profile `model` and `effort` fields and rule `why` are optional. +**Required and optional fields** + +| Field | Requirement | +| --- | --- | +| `rules` | May be absent or empty for a default-only configuration. | +| Rule `when` and `use` | Required for each rule. | +| `use` and optional top-level `default` | Accept one profile object or a non-empty array of profile objects; the single-object form remains fully backward-compatible. | +| Profile `harness` | Required in every profile. | +| Profile `model` and `effort`; rule `why` | Optional. | + +**Fields applied only by typed resolution** + Rule `approval`, `min_confidence`, and `floor`, and profile `provider` and `floor` are optional declarations that only [typed dispatch resolution](#typed-dispatch-resolution-env-typesafe_api_key) applies in code; without that opt-in they are inert, and firstmate's own intake reads them as ordinary hints. The resolver supplies the fixed neutral Choice option `No listed rule applies to this task.` for work that matches no listed rule. -`approval` accepts only `"captain"` and means a task the rule matches is never dispatched from the tool's answer alone. -`min_confidence` is a number from 0 through 1 that the rule's own probability in the answer must reach, in place of the resolver's global 0.6 floor on the answer's confidence; set it high on a rule whose wrong pick is costly and low on a rule that is a safe runner-up. -A rule `floor` names the quota-axi `provider` and `scope` whose `effectivePercentRemaining` must be at least `min_percent` for the rule's profiles to apply. -A provider-only rule floor on an expanded provider binds to its `default` account row. -An absent or unknown row or unmeasured provider makes the floor unverifiable and escalates without authorizing default routing. -A known percentage below the floor makes the tool resolve among `default` profiles instead. + +- `approval` accepts only `"captain"` and means a task the rule matches is never dispatched from the tool's answer alone. + +`min_confidence` is a number from 0 through 1. +The rule's own probability in the answer must reach it, replacing the resolver's global 0.6 floor on the answer's confidence. +Set it high when a wrong pick is costly and low when the rule is a safe runner-up. + +**Rule quota floors** + +- A rule `floor` names the quota-axi `provider` and `scope` whose `effectivePercentRemaining` must be at least `min_percent` for the rule's profiles to apply. +- A provider-only rule floor on an expanded provider binds to its `default` account row. +- An absent or unknown row or unmeasured provider makes the floor unverifiable and escalates without authorizing default routing. +- A known percentage below the floor makes the tool resolve among `default` profiles instead. + +**Provider identifiers and mappings** + A profile `provider` optionally names the quota-axi provider family whose rows apply to that profile; when present, profile and rule-floor provider IDs must match the strict whole-string pattern `^[a-z0-9]+(-[a-z0-9]+)*\z`. Bootstrap validates resolver-only `approval`, `min_confidence`, `floor`, and present `provider` values only while typed resolution is active; without the key those inert fields and the pre-existing verified-harness baseline preserve bootstrap behavior. + Typed resolution additively recognizes `gemini` because AGENTS.md section 4 verifies it for crewmate and scout dispatch. -The opted-in resolver has authoritative single-provider mappings for `claude`, `codex`, `grok`, `kimi`, `cursor`, `agy`, and `muse`; every other verified harness must declare `provider` explicitly, including multi-provider `pi`, `pi-signed`, `omp`, and `opencode` and unmapped `gemini`, `rovo`, and `devin`. -Its single-provider table is separate from the frozen legacy mapping used by `fm-quota-choose.sh`, so additions cannot alter no-key routing. -The resolver returns an actionable configuration error before any request when such a profile omits it. -A profile `floor` contains only `scope` and `min_percent`, always uses that profile's provider and matched account, and makes that one candidate ineligible below `min_percent` on the named scope. -An absent or unknown named row also makes the candidate unrankable and is reported as an unverifiable floor, not as a known shortfall. -`ultra` is native-only: the model-aware validation contract and launch mapping are owned by `bin/fm-harness.sh validate-native-effort` and `bin/fm-spawn.sh` respectively. -Codex `max` is valid when the profile selects `gpt-5.6-luna`, whose installed catalog entry supports that reasoning level. -An omitted model or effort means the selected harness uses its own default for that axis. -Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. -If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. -Except for `ultra`, which refuses unsupported profiles under the native-effort contract above, an effort value the chosen harness does not accept is recorded as `effort=` in task meta for traceability but omitted from the launch flags. -Bootstrap reports unsupported harness/model/effort combinations as a `CREW_DISPATCH` diagnostic when they are visible in the file. + +| Harness | Provider declaration on the opted-in resolver path | +| --- | --- | +| `claude`, `codex`, `grok`, `kimi`, `cursor`, `agy`, `muse` | The resolver has an authoritative single-provider mapping. | +| Every other verified harness | Must declare `provider` explicitly; this includes multi-provider `pi`, `pi-signed`, `omp`, and `opencode`, and unmapped `gemini`, `rovo`, and `devin`; omission is an actionable configuration error before any request. | + +This single-provider table is separate from the frozen legacy mapping used by `fm-quota-choose.sh`, so additions cannot alter no-key routing. + +**Profile quota floors** + +- A profile `floor` contains only `scope` and `min_percent`, always uses that profile's provider and matched account, and makes that one candidate ineligible below `min_percent` on the named scope. +- An absent or unknown named row also makes the candidate unrankable and is reported as an unverifiable floor, not as a known shortfall. + +**Model, effort, and fallback behavior** + +- `ultra` is native-only: the model-aware validation contract and launch mapping are owned by `bin/fm-harness.sh validate-native-effort` and `bin/fm-spawn.sh` respectively. +- Codex `max` is valid when the profile selects `gpt-5.6-luna`, whose installed catalog entry supports that reasoning level. +- An omitted model or effort means the selected harness uses its own default for that axis. +- Every profile array is an implicit quota-aware choice resolved through `quota-array-dispatch`. +- If no dispatch rule fits, firstmate resolves `default` through the same object-or-array path before falling back to `config/crew-harness`. +- Except for `ultra`, which refuses unsupported profiles under the native-effort contract above, an effort value the chosen harness does not accept is recorded as `effort=` in task meta for traceability but omitted from the launch flags. +- Bootstrap reports unsupported harness/model/effort combinations as a `CREW_DISPATCH` diagnostic when they are visible in the file. + See [`docs/examples/crew-dispatch.json`](examples/crew-dispatch.json) for a starting point to copy into local `config/crew-dispatch.json`; its Pi default declares the `claude` provider required for typed resolution of that Anthropic model. -When the file exists, bootstrap validates it with `jq`. -Valid files stay silent by default; with `FM_BOOTSTRAP_VERBOSE_FACTS=1`, bootstrap emits `BOOTSTRAP_INFO: crew dispatch active config/crew-dispatch.json`, one `BOOTSTRAP_INFO:` fact per rule, and one fact for the optional default profile set. -Malformed JSON, malformed rules, an empty or malformed profile array, an unverified harness, or an effort value unsupported by that harness is reported as `CREW_DISPATCH: invalid config/crew-dispatch.json - ...`. -While typed resolution is active, malformed `approval`, `min_confidence`, `floor`, and present `provider` declarations receive the same diagnostic; without the key those inert declarations preserve the pre-existing bootstrap behavior. -Missing `jq` is reported through the normal `MISSING: jq` install-consent flow. -While the file remains present, no crewmate or scout spawn may proceed without an explicit resolved harness; malformed configuration must be reported and corrected rather than selected around. + +**Validation and diagnostics** + +- When the file exists, bootstrap validates it with `jq`. +- Valid files stay silent by default; with `FM_BOOTSTRAP_VERBOSE_FACTS=1`, bootstrap emits `BOOTSTRAP_INFO: crew dispatch active config/crew-dispatch.json`, one `BOOTSTRAP_INFO:` fact per rule, and one fact for the optional default profile set. +- Malformed JSON, malformed rules, an empty or malformed profile array, an unverified harness, or an effort value unsupported by that harness is reported as `CREW_DISPATCH: invalid config/crew-dispatch.json - ...`. +- While typed resolution is active, malformed `approval`, `min_confidence`, `floor`, and present `provider` declarations receive the same diagnostic; without the key those inert declarations preserve the pre-existing bootstrap behavior. +- Missing `jq` is reported through the normal `MISSING: jq` install-consent flow. +- While the file remains present, no crewmate or scout spawn may proceed without an explicit resolved harness; malformed configuration must be reported and corrected rather than selected around. + +**Inheritance** + Secondmate homes inherit this file from the primary, so a secondmate's own crewmates apply the same dispatch profile behavior. ## Typed dispatch resolution (.env TYPESAFE_API_KEY) `bin/fm-dispatch-resolve.sh` resolves one concrete crewmate or scout profile from a written brief with typesafe.ai's System One model (Jev), so the rule match that firstmate otherwise reasons out in its own context becomes one short tool turn. It is off unless `TYPESAFE_API_KEY` is non-empty in the calling environment or the home's gitignored `.env` holds a `TYPESAFE_API_KEY=` line; the environment wins, matching the Relay and mail-plane contracts, and the Relay accessor in `bin/fm-env-lib.sh` reads the line. + Off means one `dispatch-resolve: off` line on stderr, nothing on stdout, exit 0, and no network call, so firstmate dispatches exactly as it does without the tool. This section is the single owner of the tool's operator contract; the script header owns its exact flags and output lines, and "Crew dispatch profiles" above owns the declared rule and profile fields it applies. + Rules come only from the effective home's `config/crew-dispatch.json`; `FM_CONFIG_OVERRIDE` selects the config directory for tests and specialized setup like the other scripts. ```sh bin/fm-dispatch-resolve.sh data//brief.md --project # TOON block on stdout ``` +**When firstmate invokes the resolver** + Firstmate invokes the resolve path directly after writing the brief, without a preflight; the absent-key off line is handled exactly like every other non-clear outcome. + +**What the model receives** + When on and at least one rule exists, the tool sends the project name and the brief's task-specific text as state and asks one Choice question whose options are every rule's `when` plus the fixed neutral option for no matching rule; the model never sees quota, catalogs, `why`, `use`, approvals, or confidence floors. The task-specific text is the brief's `## Captain's intent` and `## Firstmate spec` sections under `# Task` that `bin/fm-brief.sh` scaffolds, read by the same parser that feeds `fm-spawn.sh` validation and the no-mistakes `--intent` contract; a brief with neither section is sent whole. + When the sections are sent from a scout brief, the line `Brief kind: scout (report only)` comes first, taken from the scaffold's scout contract line; ship briefs and briefs sent whole get no kind line. A ship brief's delivery mode is deliberately not sent, because in live runs naming it pushed a routine ship brief toward the hardest tier (see [the verification record](verification/dispatch-resolve.md)). + The scaffold's standard setup, rules, and definition-of-done text is the same in every brief, so leaving it out keeps its safety language from reading as a signal about the task. + +**Missing or invalid rules** + An absent rules file, a default-only file, or `rules: []` returns the non-clear reason `no rules to match` without a model or quota request, leaving firstmate's existing routing in control; an existing but unreadable or malformed rules file, including a broken symlink, remains an actionable exit 2 configuration error. -Everything after the answer runs in code: the confidence floor, the matched rule's `approval` and `floor`, each candidate's `provider` and `floor`, every applicable account-wide and model/product row from one `quota-axi --json` snapshot, and the numeric `spendPriority` argmax over candidates using each candidate's limiting row. + +**Checks performed after the answer** + +After the answer, code applies all remaining checks and ranking: + +- The confidence floor and the matched rule's `approval` and `floor`. +- Each candidate's `provider` and `floor`. +- Every applicable account-wide and model/product row from one `quota-axi --json` snapshot. +- The numeric `spendPriority` argmax over candidates, using each candidate's limiting row. + The [shared quota library](../bin/fm-quota-axi-lib.sh) accepts schema 5 and schema 6 and implements the [account-matching contract](../.agents/skills/quota-array-dispatch/SKILL.md#1-eligibility). -An expanded provider with no matching account row leaves the candidate eligible but unranked. -Known applicable rows from a provider with partial quota semantics remain rankable; rows whose own status is not known remain unrankable. -A rule that declares `min_confidence` is checked against that rule's own probability, whether it is the picked option or a runner-up, so a runner-up never needs weaker support than it would as the pick. -A picked rule without `min_confidence`, and the neutral option, keep the global 0.6 floor on the answer's confidence exactly as before, so a file with no declared floors behaves as it did. -When the picked rule declares its own floor and its probability is below it, the tool takes the most probable other option whose probability clears that option's floor (a rule's `min_confidence`, otherwise 0.6), prints a `fallback:` line naming both floors, and resolves that rule as though it had been picked; no qualifying option, or two equally probable ones, is `ambiguous`. -Any applicable `exhausted_now` row or known zero bound makes that candidate ineligible, and a known profile-floor shortfall does the same before unrelated quota uncertainty is considered. -Missing or nonnumeric `spendPriority` evidence is never ranked, and every candidate is printed beside its evidence or the reason it was not rankable, including on ambiguous and approval-gated outcomes that emit no profile. -On the opted-in path, duplicate concrete profiles with the same harness, model, and effort inside one rule or the default array are configuration errors rather than ties. -The result is one of `clear` (a `profile:` line ready for `fm-spawn.sh`), `ambiguous` (confidence below the floor with no runner-up taken), `escalate` (an approval-gated rule, unverifiable rule floor, nothing rankable, or a genuine tie), or `error` (API, network, malformed response metadata, rendering, or quota-axi failure), and every one of them exits 0. -Response probabilities must contain exactly every offered choice, use numeric values from 0 through 1, and sum to approximately 1 within 0.01. -Only a usage or configuration error exits 2: an unreadable brief, an existing but unreadable or malformed canonical rules file, or missing `jq`, each reported and never selected around. -Missing `curl` is a normal structured `error` outcome with exit 0 so firstmate uses today's routing. + +- An expanded provider with no matching account row leaves the candidate eligible but unranked. +- Known applicable rows from a provider with partial quota semantics remain rankable; rows whose own status is not known remain unrankable. + +**Confidence and fallback rules** + +- A rule that declares `min_confidence` is checked against that rule's own probability, whether it is the picked option or a runner-up, so a runner-up never needs weaker support than it would as the pick. +- A picked rule without `min_confidence`, and the neutral option, keep the global 0.6 floor on the answer's confidence exactly as before, so a file with no declared floors behaves as it did. + +When the picked rule declares its own floor but its probability falls below it, the tool checks the other options: + +1. Find the most probable other option that clears its own floor: the rule's `min_confidence`, or 0.6 otherwise. +2. Print a `fallback:` line naming both floors and resolve that rule as though it had been picked. + +No qualifying option, or two equally probable qualifying options, produces `ambiguous`. + +**Candidate eligibility and evidence** + +- Any applicable `exhausted_now` row or known zero bound makes that candidate ineligible, and a known profile-floor shortfall does the same before unrelated quota uncertainty is considered. +- Missing or nonnumeric `spendPriority` evidence is never ranked, and every candidate is printed beside its evidence or the reason it was not rankable, including on ambiguous and approval-gated outcomes that emit no profile. +- On the opted-in path, duplicate concrete profiles with the same harness, model, and effort inside one rule or the default array are configuration errors rather than ties. + +**Outcomes and exit status** + +| Result | Meaning | +| --- | --- | +| `clear` | A `profile:` line ready for `fm-spawn.sh`. | +| `ambiguous` | Confidence below the floor with no runner-up taken. | +| `escalate` | An approval-gated rule, unverifiable rule floor, nothing rankable, or a genuine tie. | +| `error` | API, network, malformed response metadata, rendering, or quota-axi failure. | + +Every result above exits 0. + +- Response probabilities must contain exactly every offered choice, use numeric values from 0 through 1, and sum to approximately 1 within 0.01. +- Only a usage or configuration error exits 2: an unreadable brief, an existing but unreadable or malformed canonical rules file, or missing `jq`, each reported and never selected around. +- Missing `curl` is a normal structured `error` outcome with exit 0 so firstmate uses today's routing. + +**Firstmate retains the dispatch decision** + The tool never replaces firstmate's judgment, `quota-array-dispatch`, the captain-approval gate, or `fm-spawn.sh` validation; `AGENTS.md` section 4 owns what firstmate does with each outcome. By accepted design, a `clear` result does not enforce catalog/authentication, reasoning-class, or completion-runway gates. + Firstmate passes its profile line unless it states a reason to override, such as the brief's reasoning class or an eligible-unranked-candidate note; every non-clear result returns to the full existing intake. -The resolver and bootstrap copy an environment-provided key into a non-exported private variable and unset `TYPESAFE_API_KEY` before launching child processes, so the secret is absent from child environments. -The resolver sends the key to `curl` only as a header read from a file descriptor, never on argv, and nothing prints, logs, or writes it. -The resolver fixes the endpoint at `https://api.typesafe.ai`, model at `jev-latest`, default confidence floor at 0.6, and request timeout at 5 seconds; `TYPESAFE_API_KEY` is its only resolver-specific environment setting. +**Key handling and fixed settings** + +- The resolver and bootstrap copy an environment-provided key into a non-exported private variable and unset `TYPESAFE_API_KEY` before launching child processes, so the secret is absent from child environments. +- The resolver sends the key to `curl` only as a header read from a file descriptor, never on argv, and nothing prints, logs, or writes it. +- The resolver fixes the endpoint at `https://api.typesafe.ai`, model at `jev-latest`, default confidence floor at 0.6, and request timeout at 5 seconds; `TYPESAFE_API_KEY` is its only resolver-specific environment setting. + The live rule-match evidence is recorded in [`verification/dispatch-resolve.md`](verification/dispatch-resolve.md). ## Toolchain On session start the first mate detects what its required toolchain is missing or too old and lists each problem with either an exact install command or manual instructions. It installs automatically supported tools only after you say go; manual-only tools remain for you to install from the printed instructions. + Required tools come in two parts: a universal toolchain every home needs regardless of backend, and a per-backend delta that follows the runtime backend actually resolved for this home. -The essential universal toolchain is node, git, gh with GitHub auth via `gh auth login`, no-mistakes v1.46.0 or newer, compatible gh-axi, chrome-devtools-axi, compatible tasks-axi per "Backlog backend" above, and compatible quota-axi. + +**Universal requirements** + +Every home requires: + +- node and git. +- gh, with GitHub authentication through `gh auth login`. +- no-mistakes v1.46.0 or newer. +- Compatible gh-axi. +- chrome-devtools-axi. +- Compatible tasks-axi, as specified in "Backlog backend" above. +- Compatible quota-axi. + [`bin/fm-bootstrap.sh`](../bin/fm-bootstrap.sh) owns the axi-family floor policy and the gh-axi and lavish-axi floors, while [`bin/fm-tasks-axi-lib.sh`](../bin/fm-tasks-axi-lib.sh) and [`bin/fm-quota-axi-lib.sh`](../bin/fm-quota-axi-lib.sh) hold their own tools' floor constants. This section is the single owner of that universal toolchain list; backend guides' prerequisites point here and add only their backend-specific tools. + In that list, no-mistakes runs the validation pipeline, gh-axi and chrome-devtools-axi cover GitHub and browser operations, and tasks-axi plus quota-axi back backlog mutations and quota-aware array dispatch. Lavish is a presentation-only dependency for visual decisions and reports; nonvisual work can proceed with plain text when it is unavailable. + +**Backend requirements** + The per-backend delta is required only for the backend resolved from `FM_BACKEND`, then `config/backend`, then runtime auto-detection, then default `tmux`, so a home is never told to install a tool an inactive backend or feature would need. -That delta is owned in code by `fm_backend_required_tools` in `bin/fm-backend.sh`: the resolved backend's own session-provider CLI (`tmux`, `herdr`, `zellij`, `orca`, or `cmux`), `jq` for the JSON-emitting adapters (`herdr`, `zellij`, `cmux`) whose spawn and liveness paths parse the backend's JSON output, and the `treehouse` worktree provider for every session-provider-only backend (`tmux`, `herdr`, `zellij`, `cmux`). +`fm_backend_required_tools` in `bin/fm-backend.sh` owns the backend additions: + +| Resolved backend | Additional tools | +| --- | --- | +| `tmux` | `tmux`, `treehouse` | +| `herdr` | `herdr`, `jq`, `treehouse` | +| `zellij` | `zellij`, `jq`, `treehouse` | +| `orca` | `orca` | +| `cmux` | `cmux`, `jq`, `treehouse` | + +The JSON-emitting adapters (`herdr`, `zellij`, `cmux`) need `jq` because their spawn and liveness paths parse backend JSON. +Every session-provider-only backend (`tmux`, `herdr`, `zellij`, `cmux`) uses `treehouse` for worktrees. + Backend tool availability uses the adapter's own executable resolver, so bootstrap and spawn agree on supported non-`PATH` locations such as cmux's bundled CLI. An unknown resolved backend emits `BACKEND_INVALID` and blocks dispatch instead of silently dropping its dependency delta or falling back to tmux. + Orca provides both the task worktree and terminal endpoint (see "Runtime backend" above), so `backend=orca` requires only `orca` on top of the universal toolchain and skips both `treehouse` and every other backend's session CLI. A herdr, zellij, or cmux home is therefore never told `tmux` is missing, and the `treehouse` durable-lease upgrade check runs only for the backends that actually use treehouse. -When `config/crew-dispatch.json` exists, bootstrap also requires `jq` for dispatch profile validation. -When Relay is opted in, bootstrap also requires `curl` and `jq` before arming the relay poll shim. + +**Feature-specific requirements** + +- When `config/crew-dispatch.json` exists, bootstrap also requires `jq` for dispatch profile validation. +- When Relay is opted in, bootstrap also requires `curl` and `jq` before arming the relay poll shim. + +**Missing-tool diagnostics** + `tasks-axi` and `quota-axi` are essential bootstrap tools in every profile. -An absent or incompatible `tasks-axi` reports `MISSING: tasks-axi (install: npm install -g tasks-axi)`; when `config/backlog-backend` is not `manual`, a home with a configured non-markdown adapter or a markdown backlog refuses lifecycle mutation until compatible `tasks-axi` is on `PATH`, while a manual-backend home keeps its backlog hand-edited. -An absent or incompatible `gh-axi` reports `MISSING: gh-axi (install: npm install -g gh-axi && gh-axi setup hooks)`. -An absent or incompatible `lavish-axi` reports `PRESENTATION_UNAVAILABLE` with its required floor, install command, and explicit text fallback; [`bootstrap-diagnostics`](../.agents/skills/bootstrap-diagnostics/SKILL.md) owns the response and compatibility check before visual use. -An absent or too-old `quota-axi` reports `MISSING: quota-axi (install: npm install -g quota-axi)`; firstmate cannot resolve a profile array without a compatible binary. + +- An absent or incompatible `tasks-axi` reports `MISSING: tasks-axi (install: npm install -g tasks-axi)`; when `config/backlog-backend` is not `manual`, a home with a configured non-markdown adapter or a markdown backlog refuses lifecycle mutation until compatible `tasks-axi` is on `PATH`, while a manual-backend home keeps its backlog hand-edited. +- An absent or incompatible `gh-axi` reports `MISSING: gh-axi (install: npm install -g gh-axi && gh-axi setup hooks)`. +- An absent or incompatible `lavish-axi` reports `PRESENTATION_UNAVAILABLE` with its required floor, install command, and explicit text fallback; [`bootstrap-diagnostics`](../.agents/skills/bootstrap-diagnostics/SKILL.md) owns the response and compatibility check before visual use. +- An absent or too-old `quota-axi` reports `MISSING: quota-axi (install: npm install -g quota-axi)`; firstmate cannot resolve a profile array without a compatible binary. + +**Checkout diagnostics** + Bootstrap also reports a `TANGLE:` line when `FM_ROOT` is on a named non-default branch; follow the printed checkout remediation rather than treating it as an installable tool problem. In a read-only session that did not get the fleet lock, the same line is advisory and omits the checkout command. + +**Project refresh at startup** + The locked session-start deferred network stage runs bootstrap's best-effort project clone refresh through `fm-fleet-sync.sh`; [`fm-bootstrap.sh`'s header](../bin/fm-bootstrap.sh) owns the exact clone-refresh overlap, liveness-before-convergence, per-mate concurrency, ordered diagnostic replay, and sequential-fallback contract. -It emits `FLEET_SYNC:` for skipped refreshes that may matter, recovered self-heals, and `STUCK:` alarms. -Normal completed runs keep local-only and no-origin skips silent. -If bootstrap kills a timed-out refresh, it replays any completed `fm-fleet-sync.sh` output before the aggregate timeout skip so no finished result is lost. + +- It emits `FLEET_SYNC:` for skipped refreshes that may matter, recovered self-heals, and `STUCK:` alarms. +- Normal completed runs keep local-only and no-origin skips silent. +- If bootstrap kills a timed-out refresh, it replays any completed `fm-fleet-sync.sh` output before the aggregate timeout skip so no finished result is lost. + +**Stale Git lock recovery** + A killed refresh (or a teardown process kill) can leave an orphaned `.git/packed-refs.lock` in a clone, which makes the next refresh's fetch fail with Git's `Unable to create '...packed-refs.lock': File exists`. On that signature only, `fm-fleet-sync.sh` retries the fetch with a bounded wait for the lock to self-clear, then removes the lock and retries once more only when it can prove the lock stale, exactly like the `fm-teardown.sh` `index.lock` recovery. + It never removes a live lock, leaves any other failure shape untouched, and prints every wait, retry, and removal to stderr plus a one-line `recovered:` summary to stdout on success so that this session-start relay still surfaces the recovery. + +**Secondmate sync at startup** + The same deferred network stage performs guarded tracked-file sync and propagates declared inherited local material into each validated live home under that sequencing contract. Local routes use direct guarded filesystem operations, while remote routes delegate sync and allowlisted transfer through their configured SSH host without probing any unconfigured fleet. -It emits `SECONDMATE_SYNC:` only when a home was skipped for an actionable sync reason, inheritance failed, or a divergent shared captain-preference copy was quarantined. -When a running home advances and its loaded instruction surface (`AGENTS.md`, `bin/`, or `.agents/skills/`) changed, bootstrap sends the re-read nudge itself through the stable `fm-` selector and reports the exact completed send as `BOOTSTRAP_INFO:`. -If that send fails, bootstrap keeps an idempotent retry marker and emits `NUDGE_SECONDMATES:` with the failure reason. -The same bootstrap run emits `SECONDMATE_LIVENESS:` only when a registered secondmate is skipped or its relaunch fails; already-live and successfully relaunched secondmates are handled silently. + +- It emits `SECONDMATE_SYNC:` only when a home was skipped for an actionable sync reason, inheritance failed, or a divergent shared captain-preference copy was quarantined. +- When a running home advances and its loaded instruction surface (`AGENTS.md`, `bin/`, or `.agents/skills/`) changed, bootstrap sends the re-read nudge itself through the stable `fm-` selector and reports the exact completed send as `BOOTSTRAP_INFO:`. +- If that send fails, bootstrap keeps an idempotent retry marker and emits `NUDGE_SECONDMATES:` with the failure reason. +- The same bootstrap run emits `SECONDMATE_LIVENESS:` only when a registered secondmate is skipped or its relaunch fails; already-live and successfully relaunched secondmates are handled silently. + +**Push inherited configuration during a session** + For a mid-session inherited local-material edit where tracked-file sync is not needed, run `bin/fm-config-push.sh`. It uses the same live secondmate discovery and propagation helper as bootstrap; its [help](../bin/fm-config-push.sh) owns reporting and exit semantics, and [`fm_config_inherit_items`](../bin/fm-config-inherit-lib.sh) declares the inherited items. -When an allowlisted config item changes for an already-running local home, it sends the literal-content reread pointer described in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md); unchanged allowlisted config sends no pointer unless a previous delivery is pending. -A changed remote home instead receives one durably recorded marked re-read instruction after the allowlisted bytes have transferred because primary-local generation paths are not meaningful on another host. -The locked bootstrap inheritance pass uses the same placement-specific behavior; see `secondmate-provisioning` for the single contract owner. -That live discovery starts from `state/*.meta` records with `kind=secondmate`; `data/secondmates.md` only backfills `home=` for older or incomplete meta records. -Skipped items, such as a destination checkout that does not yet gitignore the item, are visible warnings but not hard failures. + +- When an allowlisted config item changes for an already-running local home, it sends the literal-content reread pointer described in [`secondmate-provisioning`](../.agents/skills/secondmate-provisioning/SKILL.md); unchanged allowlisted config sends no pointer unless a previous delivery is pending. +- A changed remote home instead receives one durably recorded marked re-read instruction after the allowlisted bytes have transferred because primary-local generation paths are not meaningful on another host. +- The locked bootstrap inheritance pass uses the same placement-specific behavior; see `secondmate-provisioning` for the single contract owner. +- That live discovery starts from `state/*.meta` records with `kind=secondmate`; `data/secondmates.md` only backfills `home=` for older or incomplete meta records. +- Skipped items, such as a destination checkout that does not yet gitignore the item, are visible warnings but not hard failures. ## Watched tool updates (config/watched-tools.json) @@ -671,6 +1280,7 @@ When it is present and the check is armed, [`bin/fm-tool-update-check.sh`](../bi The second condition is the reason the check exists. An update can install correctly and stay inert because an earlier `PATH` entry still holds an older copy, and a check that only asks whether a newer version is published reports that host as up to date. + The script therefore runs every copy of a watched command found on `PATH` and asks it for its own version, rather than trusting one lookup or reading a version out of a directory name. It only reports; it never installs, updates, fetches, or changes `PATH`, a version manager, or any installed tool. @@ -696,39 +1306,63 @@ This section is the single owner of the canonical schema. } ``` -Each entry needs a `name` and at least one of `command` or `git`; an entry may carry both. -A `command` entry gives the `PATH` comparison above, and adding `announce_pattern` also reports the tool's own update announcement, which is how a tool that already reports its own updates is read rather than reimplemented. -A tool does not always announce a new release on the command that prints its version: `no-mistakes --version` prints only the version, while its other commands carry the announcement. -`announce_args` names the command to search for the announcement in that case, and it is asked only of the copy `PATH` resolves; without it the version probe's own output is searched. -An `announce_pattern` that is not a usable extended regular expression stops `arm`, and during a sweep it is reported as that one tool's own check failure so one broken pattern never stops the other watched tools from being checked. -A `git` entry reports how many commits the local clone is behind its remote branch, and stays silent when the clone is current or ahead. -An omitted `branch` uses the remote's default branch, taken from the clone's own record of it and otherwise asked of the remote directly, so a `--single-branch` clone still resolves. +**Entry fields and probe behavior** + +- Each entry needs a `name` and at least one of `command` or `git`; an entry may carry both. +- A `command` entry gives the `PATH` comparison above, and adding `announce_pattern` also reports the tool's own update announcement, which is how a tool that already reports its own updates is read rather than reimplemented. +- A tool does not always announce a new release on the command that prints its version: `no-mistakes --version` prints only the version, while its other commands carry the announcement. +- `announce_args` names the command to search for the announcement in that case, and it is asked only of the copy `PATH` resolves; without it the version probe's own output is searched. +- An `announce_pattern` that is not a usable extended regular expression stops `arm`, and during a sweep it is reported as that one tool's own check failure so one broken pattern never stops the other watched tools from being checked. +- A `git` entry reports how many commits the local clone is behind its remote branch, and stays silent when the clone is current or ahead. +- An omitted `branch` uses the remote's default branch, taken from the clone's own record of it and otherwise asked of the remote directly, so a `--single-branch` clone still resolves. + Both probe kinds are read-only and bounded, and a probe that cannot answer is reported as a check failure rather than assumed current. See [`docs/examples/watched-tools.json`](examples/watched-tools.json) for a starting point to copy into local `config/watched-tools.json`. +**Arm, edit, and disarm** + Arm the check once per home with `bin/fm-tool-update-check.sh arm`. -That writes `state/tool-updates.check.sh` and binds its bytes with `bin/fm-check-register.sh`, so the existing watcher polls it on its normal cadence and turns its one line into a `check:` wake; no separate schedule is involved. -Registering the check is itself a reason to watch, so the home keeps a watcher for it after the last task is torn down, and `disarm` is what ends that need. -`bin/fm-tool-update-check.sh disarm` removes the shim, its trust binding, and the report record. -The check prints nothing when everything is current, and `state/.tool-updates` records the findings the last report was made from so the same pending update is reported once instead of on every poll. -A changed or returning condition is reported again. -Adding, removing, or changing a watched tool is an edit to this file and needs no code change or re-arming. -This file is not inherited by secondmate homes, so each home watches the tools it actually depends on. - -`FM_TOOL_UPDATE_INTERVAL` (default 900 seconds, `0` to probe on every run) sets how often probes actually run, `FM_TOOL_UPDATE_PROBE_SECS` (default 5) bounds one probe, and `FM_TOOL_UPDATE_BUDGET_SECS` (default 20) bounds a whole sweep. -A sweep that runs out of budget says which tool it did not reach rather than reporting the rest as current. -The sweep must finish inside `FM_CHECK_TIMEOUT` (default 30), because a run the watcher kills prints nothing and records nothing and would then repeat that silence on every poll. -So a budget larger than that timeout allows is cut down to what fits instead of being refused, and the cut is reported in the report line. -A budget that is not a whole number from 1 to 120 is still refused outright. + +- That writes `state/tool-updates.check.sh` and binds its bytes with `bin/fm-check-register.sh`, so the existing watcher polls it on its normal cadence and turns its one line into a `check:` wake; no separate schedule is involved. +- Registering the check is itself a reason to watch, so the home keeps a watcher for it after the last task is torn down, and `disarm` is what ends that need. +- `bin/fm-tool-update-check.sh disarm` removes the shim, its trust binding, and the report record. + +**Repeat reporting and inheritance** + +- The check prints nothing when everything is current, and `state/.tool-updates` records the findings the last report was made from so the same pending update is reported once instead of on every poll. +- A changed or returning condition is reported again. +- Adding, removing, or changing a watched tool is an edit to this file and needs no code change or re-arming. +- This file is not inherited by secondmate homes, so each home watches the tools it actually depends on. + +**Probe timing and limits** + +| Setting | Default | Purpose | +| --- | --- | --- | +| `FM_TOOL_UPDATE_INTERVAL` | 900 seconds | Time between probes; `0` probes on every run. | +| `FM_TOOL_UPDATE_PROBE_SECS` | 5 | Bounds one probe. | +| `FM_TOOL_UPDATE_BUDGET_SECS` | 20 | Bounds a whole sweep. | + +- A sweep that runs out of budget says which tool it did not reach rather than reporting the rest as current. +- The sweep must finish inside `FM_CHECK_TIMEOUT` (default 30), because a run the watcher kills prints nothing and records nothing and would then repeat that silence on every poll. +- So a budget larger than that timeout allows is cut down to what fits instead of being refused, and the cut is reported in the report line. +- A budget that is not a whole number from 1 to 120 is still refused outright. ## Mail plane (.env) The mail plane (bin/fm-mail.sh) reads unseen IMAP messages and sends one SMTP message. + +**Polling and delivery guarantees** + Its `poll` command surfaces each new message as a durable `check: mail ` wake, which is also what the standing received-mail check runs each watcher cycle. Poll emission is exactly-once-recovering: a published wake always carries a durable journal record, and a poll interrupted before recording its uid is healed from that journal, so inbound mail is never silently missed. + A duplicate wake is possible if the process is killed between the queue append and the journal write and the drain acknowledges that row before the next poll heals it, or under a triple write fault that leaves a queued row with no durable record; neither case drops mail. + +**Connection and activation** + IMAP and SMTP use implicit TLS on the default ports 993 and 465 (`IMAP4_SSL` / `SMTP_SSL`). STARTTLS and port 587 are not supported. + It is off unless the home's gitignored `.env` provides the connection values. This section is the single owner of the mail-plane configuration schema; for direct invocations, environment values override `.env`, matching the Relay contract. @@ -743,13 +1377,20 @@ FM_SMTP_HOST= # SMTP server hostname `FM_IMAP_PORT` (default 993), `FM_SMTP_PORT` (default 465), `FM_MAIL_TIMEOUT` (default 20 seconds), and `FM_MAIL_POLL_MAX_WAKES` (default 20, valid 1..200) are optional. The per-poll wake cap bounds the wakes of one `poll` run; header fetches scan a larger bounded window of new unseen uids plus already-surfaced retry-set uids, so a flood or large backlog still makes bounded progress every poll, keeping the durable wake queue bounded without ever dropping mail. + +**Unfetchable headers** + A message whose header cannot be fetched is surfaced with a degraded summary instead of being skipped, so it is never missed and cannot block later mail. A later poll retries that fetch and, on success, surfaces the real sender and subject; a persistently unfetchable message stays degraded without repeating that wake. +**Arm unattended polling** + A home that wants mail polled unattended arms the standing check in the live home: `bin/fm-mail-check.sh arm`. Arming writes `state/mail.check.sh` and registers it with the watcher's slow-check cadence (`FM_CHECK_INTERVAL`), so the plane's `poll` runs on its own: new mail still surfaces as `check: mail ` wakes from the poll, and the standing check itself also prints a line (and the watcher turns that line into a wake) unless the poll is a proven no-op. + Same-line silence is only for a proven no-op: a successful poll with no new mail, or a repeated identical pre-wake failure that cannot have queued mail. A fail-closed poll that already queued a wake, and a timeout, always print so the watcher wakes to drain it. + `FM_MAIL_CHECK_BUDGET` (default 15, valid 5..25) bounds one standing poll and is cut down to fit `FM_CHECK_TIMEOUT`. `bin/fm-mail-check.sh disarm` removes the standing check. @@ -757,135 +1398,295 @@ A fail-closed poll that already queued a wake, and a timeout, always print so th Relay lets a firstmate instance answer public mentions and act on normal reversible mention requests through firstmate's normal lifecycle. It covers both public surfaces the relay supports: `@myfirstmate` mentions on X, and mentions of the myfirstmate bot in a Discord server where it is installed. + Both surfaces are the same opt-in and the same machinery - one pairing token, one relay poll, and one reply path - so everything below applies to Discord mentions unless a line names a platform explicitly. + +**Activation, consent, and routing** + It is off unless the firstmate home's gitignored `.env` contains a non-empty `FMX_PAIRING_TOKEN`. The pairing token both identifies the relay tenant and records opt-in consent for autonomous public replies and eligible lifecycle actions. + Destructive, irreversible, or security-sensitive asks are flagged for trusted-channel confirmation instead of being executed from a public mention. The relay uses owner-only routing: a mention delivered to a home is from that home's owner/captain, while its surrounding conversation context may still include other public accounts. + +**Endpoint and environment overrides** + `FMX_RELAY_URL` is optional and defaults to `https://myfirstmate.io`, mainly for developers pointing at a local relay. For direct client invocations, environment values override `.env`; bootstrap activation still keys off `.env` presence so watcher artifacts are explicit local opt-in state. + `FMX_ENV_FILE` can point direct poll/reply client invocations at another `.env`-style file, but it does not change bootstrap activation. To turn it on: 1. Sign in at [myfirstmate.io](https://myfirstmate.io) with X or Discord. 2. For the Discord surface, use the dashboard's install link to add the myfirstmate bot to a server you administer; the X surface needs no install step. + 3. Copy the pairing token from the dashboard into this firstmate home's gitignored `.env` as `FMX_PAIRING_TOKEN=`. 4. Start a new firstmate session so bootstrap picks the token up, then mention `@myfirstmate` on X or mention the bot in a server where it is installed. The dashboard owns account creation, identity linking, bot installation, and token issuance; this document owns only what the local firstmate home does with the token once it is in `.env`. +**Generated state and watcher cadence** + The locked session-start bootstrap step turns the token into local generated state. It writes `state/x-watch.check.sh`, a byte-static identity shim for `bin/fm-x-poll.sh`, and `config/x-mode.env`, which exports `FM_CHECK_INTERVAL=30` for watcher processes in that home. + The watcher accepts the shim only when its bytes match the expected generated content, then invokes the trusted repository poll script directly instead of executing state-file source. -This section is the single owner of the Relay cadence contract: a Relay instance polls every 30 seconds instead of the default 300, only a Relay instance speeds up because a non-Relay home has no `config/x-mode.env`, and the session-start supervision operating block includes the cadence instruction when that file exists. +This section owns the Relay cadence contract: + +- A Relay instance polls every 30 seconds instead of the default 300. +- A non-Relay home has no `config/x-mode.env`, so its cadence does not change. +- When that file exists, the session-start supervision operating block includes the cadence instruction. + The active primary-harness supervision protocol owns how that sourced cadence reaches the watcher process. + +**Apply cadence changes** + Because `bin/fm-watch.sh` reads `FM_CHECK_INTERVAL` only at process start, a cadence transition - opt-in while a watcher is already running, or opt-out - is applied by restarting the home-scoped watcher through the emitted harness protocol; bootstrap deliberately never restarts the watcher itself. While a legacy daemon flag is active the daemon owns the watcher and its default cadence applies; on Pi the away-posture record alone leaves the ordinary Relay watcher cadence active, and daemon-backed Relay cadence remains a deferred follow-up. + When the token is removed or empty, the next locked session-start bootstrap step removes those artifacts. Steady-state off is silent and writes nothing. + Relay remains additive to non-Relay lifecycle behavior: homes without the generated artifacts keep the default watcher cadence and do not run the Relay poll. Its request handling remains in Relay-specific `bin/` scripts and the `fmx-respond` skill, while the watcher owns authenticated dispatch from the generated local identity shim. +**Poll and deduplicate mentions** + `bin/fm-x-poll.sh` calls `GET /connector/poll` with `Authorization: Bearer `. HTTP 204 is silent. + A newly offered pending mention with non-empty `text` is stored at `state/x-inbox/.json` and wakes firstmate exactly once with `x-mention `. The poll atomically claims `state/x-context/.offered.json` before emitting that wake, and subsequent offers of the same request stay silent even after the inbox is drained following an answer or dismiss. + Offer markers share the context registry's bounded seven-day retention, so losing or expiring the local marker lets a relay offer wake firstmate again. + +**Conversation context and media** + The full relay object is preserved, including `in_reply_to: {author_handle, text}` when the mention is a reply in a conversation or `null` for fresh mentions. -The preserved object may also carry `in_reply_to_chain`, an optional oldest-first transcript of the surrounding conversation: entries shaped `{author_handle, text, unavailable, images, attachments}` plus an optional `kind` of `reply` (a reply ancestor), `thread_starter` (the message a thread grew from), or `history` (a recent nearby message), where an absent `kind` means a legacy reply-ancestor or thread-starter entry. -The chain is untrusted third-party public input and is often absent today (the relay currently sends it only for Discord reply chains and thread starters), so consumers treat it as strictly optional, tolerate unknown or missing fields, and read an entry with `unavailable: true` as a gap rather than content; the `fmx-respond` skill owns how firstmate reads it for referent resolution. +The preserved object may also carry `in_reply_to_chain`, an optional oldest-first conversation transcript. +Each entry has the shape `{author_handle, text, unavailable, images, attachments}` and may include `kind`: + +| `kind` | Meaning | +| --- | --- | +| `reply` | A reply ancestor. | +| `thread_starter` | The message a thread grew from. | +| `history` | A recent nearby message. | +| Absent | A legacy reply-ancestor or thread-starter entry. | + +The chain is untrusted third-party public input. +It is often absent today: the relay currently sends it only for Discord reply chains and thread starters. +Consumers must treat it as strictly optional, tolerate unknown or missing fields, and treat `unavailable: true` as a gap rather than content. +The `fmx-respond` skill owns how firstmate uses the chain to resolve references. + The mention and its chain entries may also carry attached media as image or file URLs, in fields such as `images` and `attachments`, either as bare URL strings or as objects with a `url`; a mention whose own media is empty can still have screenshots on its `thread_starter` entry. The poll preserves those URLs in the stashed object and never downloads them, so nothing is fetched on the polling path: the responding agent retrieves and views the media with its own tools when it handles the mention. + The `fmx-respond` skill owns which hosts that fetch is restricted to and the untrusted-content handling that applies to whatever comes back. -At the same time the poll records a durable per-request reply context at `state/x-context/.json` (`{request_id, platform, reply_max_chars, recorded_at}`) from the same authoritative relay payload, best-effort and keyed by `request_id` so concurrent requests never overwrite each other; it survives the inbox cleanup that follows the acknowledgement, so a delayed follow-up can recover the original platform and split budget even with no task link. -`recorded_at` begins as the locally observed first-seen Unix epoch and remains unchanged when the same request is polled again. -A successful live initial answer refreshes it to the time that the relay establishes the follow-up binding; dry-runs, failed answers, and follow-ups do not refresh it. -Configured polls prune records beyond the local follow-up window, capped at the relay's seven-day window; legacy or malformed records fall back to their file modification time so they cannot remain indefinitely. -The record is written only when a platform or explicit budget is actually known, so an unknown-platform mention leaves no useless entry. + +**Durable reply context** + +The same authoritative relay payload also supplies durable per-request reply context at `state/x-context/.json`, with shape `{request_id, platform, reply_max_chars, recorded_at}`. +The poll writes this best-effort record keyed by `request_id`, so concurrent requests never overwrite each other. +It survives inbox cleanup after acknowledgement, allowing a delayed follow-up to recover the original platform and split budget even without a task link. + +- `recorded_at` begins as the locally observed first-seen Unix epoch and remains unchanged when the same request is polled again. +- A successful live initial answer refreshes it to the time that the relay establishes the follow-up binding; dry-runs, failed answers, and follow-ups do not refresh it. +- Configured polls prune records beyond the local follow-up window, capped at the relay's seven-day window; legacy or malformed records fall back to their file modification time so they cannot remain indefinitely. +- The record is written only when a platform or explicit budget is actually known, so an unknown-platform mention leaves no useless entry. + +**Handle requests and acknowledgements** + The `fmx-respond` skill decides whether the stashed mention is an actionable request, a question, or a pure acknowledgment. -Actionable reversible requests are run through intake, backlog, dispatch, investigation, or ship flow as appropriate. -If the work completes in that turn, the public reply reports the outcome. -If the request spawns a longer-running task, firstmate posts an acknowledgement through the normal answer endpoint, links the task to the mention with `bin/fm-x-link.sh`, and posts up to three completion follow-ups on genuine milestones, finishing with a `--final` one for ordinary Relay-linked work. When a typed promised-final commitment is registered, `bin/fm-public-followup.sh` owns the terminal reply and clears the legacy link after its receipt is validated. -That link stores optional reply-platform context so Discord-originated follow-ups keep Discord's larger message budget after the inbox file has been drained. + +- Actionable reversible requests are run through intake, backlog, dispatch, investigation, or ship flow as appropriate. +- If the work completes in that turn, the public reply reports the outcome. +- If the request spawns a longer-running task, firstmate posts an acknowledgement through the normal answer endpoint, links the task to the mention with `bin/fm-x-link.sh`, and posts up to three completion follow-ups on genuine milestones, finishing with a `--final` one for ordinary Relay-linked work. + When a typed promised-final commitment is registered, `bin/fm-public-followup.sh` owns the terminal reply and clears the legacy link after its receipt is validated. +- That link stores optional reply-platform context so Discord-originated follow-ups keep Discord's larger message budget after the inbox file has been drained. + +**Resolve the reply platform and budget** + Platform/budget resolution is layered and independent of the task link: a per-axis `FMX_REPLY_PLATFORM` / `FMX_REPLY_MAX_CHARS` override (how `bin/fm-x-followup.sh` passes a recorded link's context) wins. -For either axis without an override, `bin/fm-x-lib.sh:fmx_resolve_reply_context` owns the source order: the durable per-request registry is consulted first, then the still-present inbox payload, then - for a follow-up posted live by request_id - an authoritative relay lookup via `POST /connector/request-context` (`{request_id}` in, `{platform, reply_max_chars}` back). +For either axis without an override, `bin/fm-x-lib.sh:fmx_resolve_reply_context` consults these sources in order: + +1. The durable per-request registry. +2. The still-present inbox payload. +3. For a follow-up posted live by request_id only, an authoritative relay lookup through `POST /connector/request-context`: `{request_id}` in, `{platform, reply_max_chars}` back. + This is what keeps a delayed request-id follow-up on the original platform's budget even after the inbox is drained and with no task link surviving; the relay step is confined to the live follow-up path so the answer path and every dry-run stay network-free. -The link is home-local by construction, because it lives in that home's own `state/.meta`: work routed to a secondmate has no record here, so `bin/fm-x-link.sh` refuses it, names the registered secondmate home the task was found in when it can, and points at the promised-final path (`bin/fm-public-followup.sh register ... --work-home secondmate:`), which is the only follow-up mechanism that binds work in another home. -`bin/fm-x-link.sh` follows the same ordering when recording a fresh link's context and requires `jq`; its request-context lookup is best-effort: no token or `curl`; a non-2xx response; an unresolved response; or a relay version without that endpoint leaves the context unknown. -In that case the link is still recorded but `bin/fm-x-link.sh` prints a loud warning; and when either a follow-up's platform or explicit budget cannot be authoritatively resolved from any source, `bin/fm-x-reply.sh` refuses it (fail-safe exit 8) rather than posting with a local default - firstmate holds and retries it once both values are recoverable. + +**Link tasks and handle missing context** + +The link lives in the current home's `state/.meta`. +Work routed to a secondmate has no record here, so `bin/fm-x-link.sh` refuses to link it. +When possible, the refusal names the registered secondmate home containing the task. + +It also points to `bin/fm-public-followup.sh register ... --work-home secondmate:`. +This promised-final path is the only follow-up mechanism that binds work in another home. +`bin/fm-x-link.sh` uses the same order when recording a fresh link's context and requires `jq`. +Its request-context lookup is best-effort. +Any of these conditions leaves the context unknown: + +- No token or `curl`. +- A non-2xx response. +- An unresolved response. +- A relay version without that endpoint. + +The link is still recorded, but `bin/fm-x-link.sh` prints a loud warning. +If either the follow-up platform or explicit budget cannot be authoritatively resolved from any source, `bin/fm-x-reply.sh` refuses with fail-safe exit 8. +Firstmate holds the follow-up and retries once both values are recoverable; it never posts with a local default. + +**Carry a link to a successor task** + Fresh links start with `x_followups=0` and the current timestamp; when relinking the same relay request onto a successor task, pass paired `--carry-count --carry-ts ` flags plus any prior `x_platform=` and `x_reply_max_chars=` as `--carry-platform --carry-max ` so the successor preserves the already-consumed follow-up count, original 7-day window, and reply split budget. + +**Dismiss mentions** + Pure acknowledgments or mentions with nothing to answer are dismissed through `bin/fm-x-dismiss.sh` before the local inbox file is cleared. Dismiss sends `POST /connector/dismiss` with `{request_id}`, posts no text, and tells the relay to drop the request instead of re-offering it or falling back to an offline auto-reply; on success it clears that request's durable reply-context record, while the separate offer marker remains for its bounded retention so a brief relay re-offer stays silent. + +**Poll errors** + Relay auth or config problems are reported once as `x-mode-error ...` until recovery. A failed durable offer claim is likewise reported once as `x-mode-error cannot record mention offer` and remains deduplicated through quiet no-pending polls until a later offer confirms an existing valid marker or claims a new one. + +**Post replies and follow-ups** + Live replies are posted by `bin/fm-x-reply.sh`, which sends `POST /connector/answer` with `{request_id,text}` for one-message replies. Add `--image ` to attach one local PNG, JPEG, GIF, WebP, BMP, or TIFF as `{media_type,data_base64}` in the relay's optional `image` object. + Completion follow-ups use `bin/fm-x-followup.sh`, which checks the local `state/.meta` link and sends the same payload shape through `POST /connector/followup` by calling `bin/fm-x-reply.sh --followup`, up to three times per link within the window. Add `--image ` there too when a completion follow-up should carry an image. -A successful post increments the local `x_followups=` counter and keeps the link, unless `--final` was passed or the new count reaches the cap, in which case the link is cleared instead; a failed post leaves the link and counter untouched so it can be retried. -The relay itself rejects a follow-up past its own cap or window with HTTP 409 and may include `{"error":"followup_unavailable"}` in the response body; the client surfaces any follow-up 409 as a distinguishable exit code and uses the body marker only for a sharper diagnostic. -`fm-x-followup.sh` treats that exit exactly like a locally-detected expiry - clearing the link and skipping quietly rather than retrying - so an older single-follow-up relay or an already-exhausted binding degrades gracefully. -It treats `fm-x-reply.sh`'s fail-safe refusal (exit 8: platform or explicit budget unresolved) differently: that is a retryable hold, so the link is KEPT and the follow-up is retried once both values can be recovered, never posted with a local default. -Past-window relay rejections are only guaranteed while the expired binding row still exists on the relay side; after its cleanup sweep, a very-late follow-up call may instead see a benign no-op 200, which is why the local window and cap pruning remains the primary guard. -Reply splitting is platform-aware: an explicit relay platform field (`reply_platform`, `platform`, `target_platform`, `source_platform`, or `provider`) wins, otherwise a legacy `tweet_id` beginning with `discord:` selects Discord and a numeric `tweet_id` selects X. -An explicit relay limit field (`reply_max_chars`, `reply_max_characters`, `message_max_chars`, `message_limit`, or `max_chars`) wins over the platform defaults. -If the reply exceeds the selected budget, the client splits it into a numbered thread on fenced-code, paragraph, line, and word boundaries and sends `{request_id,text,texts}`, where `texts` is the ordered chunk list and `text` remains the first chunk for older relays. -When `--image ` is present on a split reply, the image rides the first/opener message and later chunks stay text-only. -`FMX_X_REPLY_MAX_CHARS` defaults to 280 and clamps to a minimum of 50; `FMX_DISCORD_REPLY_MAX_CHARS` defaults to 1900, clamps to a minimum of 50, and resets values above Discord's 2000-character limit back to 1900. -`FMX_X_THREAD_MAX` defaults to 25 and caps oversized reply threads for every platform, marking the last retained message with an ellipsis when truncation is needed. -`FMX_FOLLOWUP_MAX_AGE_SECS` defaults to 604800 (7 days) and controls the local completion follow-up window; `FMX_FOLLOWUP_MAX_COUNT` defaults to 3 and controls the local follow-up cap. + +**Follow-up success, expiry, and retry** + +- A successful post increments the local `x_followups=` counter and keeps the link, unless `--final` was passed or the new count reaches the cap, in which case the link is cleared instead; a failed post leaves the link and counter untouched so it can be retried. +- The relay itself rejects a follow-up past its own cap or window with HTTP 409 and may include `{"error":"followup_unavailable"}` in the response body; the client surfaces any follow-up 409 as a distinguishable exit code and uses the body marker only for a sharper diagnostic. +- `fm-x-followup.sh` treats that exit exactly like a locally-detected expiry - clearing the link and skipping quietly rather than retrying - so an older single-follow-up relay or an already-exhausted binding degrades gracefully. +- It treats `fm-x-reply.sh`'s fail-safe refusal (exit 8: platform or explicit budget unresolved) differently: that is a retryable hold, so the link is KEPT and the follow-up is retried once both values can be recovered, never posted with a local default. +- Past-window relay rejections are only guaranteed while the expired binding row still exists on the relay side; after its cleanup sweep, a very-late follow-up call may instead see a benign no-op 200, which is why the local window and cap pruning remains the primary guard. + +**Split replies by platform** + +- Reply splitting is platform-aware: an explicit relay platform field (`reply_platform`, `platform`, `target_platform`, `source_platform`, or `provider`) wins, otherwise a legacy `tweet_id` beginning with `discord:` selects Discord and a numeric `tweet_id` selects X. +- An explicit relay limit field (`reply_max_chars`, `reply_max_characters`, `message_max_chars`, `message_limit`, or `max_chars`) wins over the platform defaults. +- If the reply exceeds the selected budget, the client splits it into a numbered thread on fenced-code, paragraph, line, and word boundaries and sends `{request_id,text,texts}`, where `texts` is the ordered chunk list and `text` remains the first chunk for older relays. +- When `--image ` is present on a split reply, the image rides the first/opener message and later chunks stay text-only. + +**Reply and follow-up limits** + +| Setting | Default | Limit or behavior | +| --- | --- | --- | +| `FMX_X_REPLY_MAX_CHARS` | 280 | Clamps to a minimum of 50. | +| `FMX_DISCORD_REPLY_MAX_CHARS` | 1900 | Clamps to a minimum of 50; values above Discord's 2000-character limit reset to 1900. | +| `FMX_X_THREAD_MAX` | 25 | Caps oversized reply threads on every platform; truncation marks the last retained message with an ellipsis. | +| `FMX_FOLLOWUP_MAX_AGE_SECS` | 604800 (7 days) | Local completion follow-up window. | +| `FMX_FOLLOWUP_MAX_COUNT` | 3 | Local follow-up cap. | + +**Preview with dry-run** Set `FMX_DRY_RUN` to preview replies and dismissals without posting. Truthy means anything except unset, empty, `0`, `false`, `no`, or `off`; an explicit environment value wins over `.env`. -In dry-run, `fm-x-reply.sh` records the would-be payload to `state/x-outbox/.json`, including `texts` for a thread and an `endpoint` marker for follow-up previews, prints a `DRY RUN` summary to stderr, echoes the `request_id`, and exits 0. -When an image is attached, the dry-run record uses compact `{media_type, bytes, source_path}` metadata instead of writing the base64 bytes. -In dry-run, `fm-x-dismiss.sh` records `{request_id, endpoint:"dismiss"}` to the same outbox path, prints a `DRY RUN` summary, echoes the `request_id`, and exits 0. -The live answer and follow-up bodies intentionally stay the same shape, including optional `image`; the relay distinguishes them by endpoint, and dismiss stays `{request_id}`. -These paths need `jq` to build the JSON payload, but they run before token and network checks, so they need neither `FMX_PAIRING_TOKEN` nor `curl`. + +- In dry-run, `fm-x-reply.sh` records the would-be payload to `state/x-outbox/.json`, including `texts` for a thread and an `endpoint` marker for follow-up previews, prints a `DRY RUN` summary to stderr, echoes the `request_id`, and exits 0. +- When an image is attached, the dry-run record uses compact `{media_type, bytes, source_path}` metadata instead of writing the base64 bytes. +- In dry-run, `fm-x-dismiss.sh` records `{request_id, endpoint:"dismiss"}` to the same outbox path, prints a `DRY RUN` summary, echoes the `request_id`, and exits 0. +- The live answer and follow-up bodies intentionally stay the same shape, including optional `image`; the relay distinguishes them by endpoint, and dismiss stays `{request_id}`. +- These paths need `jq` to build the JSON payload, but they run before token and network checks, so they need neither `FMX_PAIRING_TOKEN` nor `curl`. ### Promised public replies (state/public-followup) A relay request that spawns real work can leave firstmate owing a specific public reply in a specific thread. That promise is a typed `kind=public-followup` obligation whose state machine is owned entirely by `tasks-axi public-followup`, while the full private conversation context stays only in `state/x-context/`. + Firstmate's bounded registration retains the obligation's public-safe request binding so a delivered loop can be rechained without the original inbox. `bin/fm-public-followup.sh` is firstmate's side: it registers a commitment, reconciles typed terminal work results into it, posts the final reply through `bin/fm-x-reply.sh --followup`, and explicitly rechains or retires the retained loop. + Run `bin/fm-public-followup.sh --help` for the exact subcommands and flags. -Registration is what creates this home's private transport under `state/public-followup/` (mode 0700): `registry/` for the bounded private binding of each open public loop (the record survives delivery, stamped `state=delivered`, and is removed only by `retire`), `events/` for typed terminal results awaiting reconciliation, `consumed/` for the accepted-event ledger, `rejected/` for refusals kept with a one-line reason, `rejection-wakes/` for each refusal's not-yet-raised wake, `retired/` for the mode-0600 reason-and-time receipt written before removal, and `surfaced` for the poll's last-surfaced signature. -A work home that reports across a machine boundary also gets `outbox/`, described below. +**Private transport records** + +Registration creates this home's private transport under `state/public-followup/` with mode 0700: + +| Entry | Purpose and retention | +| --- | --- | +| `registry/` | Bounded private binding for each open public loop; survives delivery with `state=delivered`; only `retire` removes it. | +| `events/` | Typed terminal results awaiting reconciliation. | +| `consumed/` | Accepted-event ledger. | +| `rejected/` | Refusals retained with a one-line reason. | +| `rejection-wakes/` | Each refusal's not-yet-raised wake. | +| `retired/` | Mode-0600 reason-and-time receipt written before removal. | +| `surfaced` | The poll's last-surfaced signature. | +| `outbox/` | Also created in a work home that reports across a machine boundary; described below. | + +**Which home posts the reply** + The home that owns the commitment also owns the outward post, because only it holds the relay consent, the request context, and the opaque thread binding. Work routed elsewhere reports a typed terminal result with `bin/fm-public-followup-emit.sh` and never looks for the thread; when writing directly into the owning home, that emitter refuses a home with no registration for the named obligation. + +**Prepare and validate terminal results** + `bin/fm-public-followup.sh brief` pre-fills every deliverable value the binding determines, such as `report_path=data//report.md`, and states the accepted format of every value it cannot know. -The emitter validates deliverable values and known required keys before publishing, including the relative `report_path` format, and names correctable mistakes at the work home. -A direct emit reads the obligation from `tasks-axi`; a staged emit cannot read that remote record, so `brief` supplies its required keys in the printed command. -If those flags are omitted from a staged command, it still checks values but cannot detect missing keys until the owning home's `consume` rejects the event and queues a rejection wake. -The [emitter header](../bin/fm-public-followup-emit.sh) and its `--help` own the exact flags and outcome-dependent validation rules. + +- The emitter validates deliverable values and known required keys before publishing, including the relative `report_path` format, and names correctable mistakes at the work home. +- A direct emit reads the obligation from `tasks-axi`; a staged emit cannot read that remote record, so `brief` supplies its required keys in the printed command. +- If those flags are omitted from a staged command, it still checks values but cannot detect missing keys until the owning home's `consume` rejects the event and queues a rejection wake. +- The [emitter header](../bin/fm-public-followup-emit.sh) and its `--help` own the exact flags and outcome-dependent validation rules. + +**Clear legacy links in remote homes** + When that work lives in a REMOTE secondmate home, delivery clears its bound legacy link after validating the public receipt, while retirement clears the link before closing the loop, and both clears run over that route's SSH transport. -Readable remote state that proves no link exists succeeds without a write, while a present link is cleared only when its Relay request identity matches the registration and the state is writable; an identity mismatch, unreadable or unsafe state, an unavailable write or lock, an older remote copy, or a host that never confirms the clear leaves the loop retained for reconciliation. +Readable remote state proving that no link exists succeeds without a write. +A present link is cleared only when its Relay request identity matches the registration and the state is writable. +Any of these conditions retains the loop for reconciliation: + +- An identity mismatch. +- Unreadable or unsafe state. +- An unavailable write or lock. +- An older remote copy. +- A host that never confirms the clear. + +**Duplicate and failed results** + A terminal event's id is derived from its identity tuple, so a duplicate report, a retry, or a replay after restart resolves to the same event and changes nothing. When bound work ends failed or parked, its typed failed result remains deliverable even when the promised final expected a merged pull request, so the owed reply carries the honest failure instead of remaining stranded. +**Collect results across machines** + Work bound to a REMOTE secondmate home reports across a machine boundary, where no local path reaches the owning home. -`bin/fm-public-followup.sh brief` therefore prints that worker the route's own code root and home with `--stage-in`, so the typed result is staged in `outbox/` in the home where the work actually runs rather than written to a path that only exists on the owning machine. -The owning home collects staged results for open registrations over the same SSH route it reaches that secondmate on, because that transport only runs in the outbound direction: `consume` pulls them into its own `events/` and then reconciles them exactly as it reconciles a local report. -Non-open registrations owe no result, so `consume` skips them without contacting their routes; an open registration whose reachable route has nothing staged remains pending without an error. -Collection is non-destructive until the result is durably held, and the staged copy is retired only afterwards, so a dropped connection can never lose a terminal result. -For an open registration, a work home that cannot be reached is named in `consume`'s output and keeps the promise open; it is never reported as an empty inbox. + +- `bin/fm-public-followup.sh brief` therefore prints that worker the route's own code root and home with `--stage-in`, so the typed result is staged in `outbox/` in the home where the work actually runs rather than written to a path that only exists on the owning machine. +- The owning home collects staged results for open registrations over the same SSH route it reaches that secondmate on, because that transport only runs in the outbound direction: `consume` pulls them into its own `events/` and then reconciles them exactly as it reconciles a local report. +- Non-open registrations owe no result, so `consume` skips them without contacting their routes; an open registration whose reachable route has nothing staged remains pending without an error. +- Collection is non-destructive until the result is durably held, and the staged copy is retired only afterwards, so a dropped connection can never lose a terminal result. +- For an open registration, a work home that cannot be reached is named in `consume`'s output and keeps the promise open; it is never reported as an empty inbox. + Run `bin/fm-public-followup-collect.sh --help` for the staged-result commands the owning home runs over that route. +**Activation and idle cost** + Activation is the same `.env` `FMX_PAIRING_TOKEN` contract as the rest of Relay, with no second flag. -A home without that token runs one file test and stops: no `tasks-axi` call, no backlog or request-context scan, and no `state/public-followup/` directory. -Ordinary startup, polling, cleanup, and silent read-side subcommands also produce no output; commands that require an active relay report that configuration error after the same gate. -A relay-enabled home with no registered commitment stops at an O(1) directory presence check, so the empty state costs no CLI call and adds no periodic scan. + +- A home without that token runs one file test and stops: no `tasks-axi` call, no backlog or request-context scan, and no `state/public-followup/` directory. +- Ordinary startup, polling, cleanup, and silent read-side subcommands also produce no output; commands that require an active relay report that configuration error after the same gate. +- A relay-enabled home with no registered commitment stops at an O(1) directory presence check, so the empty state costs no CLI call and adds no periodic scan. + +**Wake on new or rejected results** + Unreconciled terminal results ride the existing 30-second relay poll rather than a new process or timer: `bin/fm-x-poll.sh` compares the pending-event signature against `surfaced` and wakes firstmate once per new result set. -A terminal event `tasks-axi` refuses during `consume` is quarantined with a reason naming the specific deliverable, outcome, or missing key where one is identifiable, and the same poll wakes the owning home with a `public-followup rejected ...` line carrying that reason. -The refused event stays pending until that wake is recorded, and a queued wake survives a failed read or write to poll output. -That makes the wake at-least-once rather than exactly-once: a cleanup that fails after the line was already raised - a wake directory that cannot be written, or a refused event that could not be drained - raises the same refusal again on a later poll. -A repeat carries the same event id and the same reason as the quarantined rejection, which is how an already-handled refusal is recognized. -Acknowledge it without re-acting; re-emitting an already accepted corrected result is harmless but redundant because its derived event id is already in the accepted ledger. + +- A terminal event `tasks-axi` refuses during `consume` is quarantined with a reason naming the specific deliverable, outcome, or missing key where one is identifiable, and the same poll wakes the owning home with a `public-followup rejected ...` line carrying that reason. +- The refused event stays pending until that wake is recorded, and a queued wake survives a failed read or write to poll output. +- That makes the wake at-least-once rather than exactly-once: a cleanup that fails after the line was already raised - a wake directory that cannot be written, or a refused event that could not be drained - raises the same refusal again on a later poll. +- A repeat carries the same event id and the same reason as the quarantined rejection, which is how an already-handled refusal is recognized. +- Acknowledge it without re-acting; re-emitting an already accepted corrected result is harmless but redundant because its derived event id is already in the accepted ledger. + +**Startup, teardown, and retries** + The session-start digest separately prints a "Public commitments" subsection from disk when, and only when, this home is relay-active and still holds an open public loop (a reply still owed, or a delivered loop with nothing owed), so compaction and restart are non-events. `bin/fm-teardown.sh` refuses to clean up a task while this home still owes a public reply for exactly that work, unless `--force` carries explicit discard approval. + `FM_PF_RETRY_BACKOFF_SECS` (default 900) sets the next-attempt time recorded with a retryable delivery error. See [verification/public-followup.md](verification/public-followup.md) for the current maintainer evidence behind restart recovery, failed terminal outcomes, retained-loop disposition, and the relay-disabled zero-overhead guarantee. @@ -893,18 +1694,34 @@ See [verification/public-followup.md](verification/public-followup.md) for the c A home can explicitly enable a trusted external `process-event-adapter/1` package without adding package code to Firstmate. This is one narrow extension type, not a general plugin or hook system. + [`extension-bindings.md`](extension-bindings.md) owns the manifest, binding, trust, handshake, invocation-envelope, capability, version-compatibility, and authority-boundary contracts. `bin/fm-extension.sh --help` and `bin/fm-procevent.sh --help` own exact command mechanics. +**Discovery and disabled behavior** + Discovery reads only mode-`0600` bindings under this home's mode-`0700` `config/extensions.d/` directory. The current directory, projects, task copies, worker text, environment payloads, and Pi packages are never searched for extensions. + When the directory is absent, ordinary process-event commands perform only a bounded absence check, create no package or extension state, and preserve every built-in adapter path. +**Bind a trusted package** + Binding separates the package's own manifest from this home's explicit enablement. -`bind` validates the source package, computes every digest, copies the complete tree into the read-only content-addressed `data/extensions/packages/` store, performs the live handshake, and atomically publishes the enabled adapter-name subset. +`bind` performs these steps: + +1. Validate the source package and compute every digest. +2. Copy the complete tree into the read-only content-addressed `data/extensions/packages/` store. +3. Perform the live handshake. +4. Atomically publish the enabled adapter-name subset. + The operator supplies trust and required consent facts, not hashes. + +**Working state and cleanup** + `state/extensions//` is created when binding performs its initial handshake and is that package's home-local working namespace for later verification and invocation. `state/extension-invocations/` contains private host-owned exact process-group cleanup records only while an enabled package invocation is starting or running; retirement and reconciliation retain their existing owners until those records prove the group extinct. + This integrity boundary does not sandbox trusted same-user code, so bind only a package trusted to run with the operator's operating-system access. The shipped `file-signal` package is a complete neutral example. @@ -926,6 +1743,9 @@ bin/fm-extension.sh verify org.firstmate.example.file-signal Use an absent destination for the copy so the source identity remains inspectable and reproducible. For a non-default home, set `FM_HOME=` on every command; local and remote secondmate homes bind the package independently, and bindings are not inherited. + +**Bind on a remote secondmate** + For a configured remote secondmate, keep the package at the controller and transfer it through the authenticated `fm-on` route: ```sh @@ -938,10 +1758,16 @@ bin/fm-extension.sh remote-bind \ The command serializes only the validated extension package, stages it below the addressed remote home's fixed extension staging root, binds it there, and prints transfer and binding digests. Registration uses `bin/fm-on.sh fm-procevent.sh ...`. + +**Retire a binding** + After retiring every registration with its printed owner token and handling every captured result, retire the enabled remote binding and its exact staged transfer together with `bin/fm-on.sh fm-extension.sh retire-transfer --if-transfer-digest --if-binding-digest `. For a direct local binding, use `bin/fm-extension.sh retire-binding --if-binding-digest ` after the same process-event retirement and handling steps. + Both commands retain the retired identity reversibly and leave unrelated bindings and content-addressed installed packages unchanged. +**Register a completion source** + Register one file completion source with a path-safe source id and an explicit non-secret source configuration reference. Credential values never belong in that reference, command argv, or a process-event result: @@ -952,195 +1778,393 @@ bin/fm-procevent.sh reconcile ``` `register-extension` prints the new registration's owner token and exact owner-matched retirement command. + +**Classify and acknowledge results** + The source waits outside the conversational turn, and its completed result arrives through the existing process-event `check` path. Classify the captured result through its immutable package identity with `bin/fm-procevent.sh classify `, acknowledge it with the existing `handled` command only after it is handled, and use the printed `retire --if-owner` command when explicit retirement is needed. + +**Keep blocking sources out of the turn** + Never run the registered blocking source command directly in a conversational turn. ## Process-to-event sources (state/procevent) A long-polling external process is registered as a *source* through its adapter, whose header and `--help` own the commands and flags. `bin/fm-procevent.sh` owns the generic contract; built-in adapters retain their tracked `bin/fm-procevent-.sh` commands, while an explicitly bound external adapter routes through the trusted host contract above. + `bin/fm-procevent-lavish.sh` is the first built-in adapter and wraps only the currently published `lavish-axi poll` interface. + +**Open the Lavish artifact first** + Before arming any Lavish source, open its artifact with `lavish-axi` so the saved session identifies the board's server; each poll attempt derives its host and port from that session and refuses missing or invalid session evidence before consuming a staged worker reply. + +**Retry interrupted Lavish polls** + That adapter, and only that adapter, retries the one exact transient response a cut-short listener returns while its marks remain available (`error: Lavish Editor poll response was interrupted` with `code: SERVER_ERROR`), up to 12 times with poll starts at least 5 seconds apart, so an internal retry never reaches the runner as a captured result. This start-to-start governor is a no-op after a normally blocking poll but caps an immediately returning poll under the shipped defaults independently of the owner lease and registration launch pacing. + Real feedback, ended and missing sessions, any other `SERVER_ERROR`, and that same interruption still standing once the bound is spent are all captured and announced normally; `FM_LAVISH_POLL_RETRY_DELAY` is a bounded 1 to 60 second test override for the interval only, and the runner itself stays adapter-agnostic. An already-armed Lavish source keeps its registered listener command until it is retired and armed again, so retire the source, then arm it again to adopt this retry policy. ### Crew-hosted Lavish review boards +**Arm and confirm a listener** + A live task that hosts a Lavish board owns its listener, so firstmate must never arm that board. After opening the artifact as required above, the worker arms it with `bin/fm-procevent-lavish.sh arm --for ` and never runs `lavish-axi poll` itself. + `arm` prints `armed` only after the process-event owner confirms this registration generation's listener is running, and otherwise returns nonzero without that line. -The confirmation is the same live claim or launch-stamp evidence `reconcile` already uses, bounded by `FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS`, and a failed confirmation retires a source that never started unless `retire` refuses because something may still own it, in which case the registration stays for `reconcile` or a human. -An earlier registration's listener that releases the board inside the confirm window lets the new registration start, and `arm` then reports `armed` as usual. -When a live listener from an earlier registration of the same board still holds it when the window ends, `arm` exits zero with `still-listening` instead of `armed`, because that earlier listener keeps serving the board and the new registration takes effect only after the source is retired and armed again. -The arm is refused unless that task id has valid, identity-matching endpoint metadata, because a board whose owner has no endpoint would collect feedback nobody can be told about. + +- The confirmation is the same live claim or launch-stamp evidence `reconcile` already uses, bounded by `FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS`, and a failed confirmation retires a source that never started unless `retire` refuses because something may still own it, in which case the registration stays for `reconcile` or a human. +- An earlier registration's listener that releases the board inside the confirm window lets the new registration start, and `arm` then reports `armed` as usual. +- When a live listener from an earlier registration of the same board still holds it when the window ends, `arm` exits zero with `still-listening` instead of `armed`, because that earlier listener keeps serving the board and the new registration takes effect only after the source is retired and armed again. +- The arm is refused unless that task id has valid, identity-matching endpoint metadata, because a board whose owner has no endpoint would collect feedback nobody can be told about. + +**Acknowledge a round by re-arming** + The registration persists as one task-owned source record, while each captured nonterminal round remains open until the worker re-arms and the existing handled marker acknowledges that round. Re-arm is that acknowledgement and nothing else: the board is armed once while no record exists, and a further arm by the same owner is refused unless an unacknowledged nonterminal round is waiting, so a generation already carrying a reply is never replaced before its listener posts it. -Re-arm never acquires, releases, or hands off the source claim, and it may carry `--agent-reply-file ` whose contents are copied into that generation's own private staging file and handed once to the published `--agent-reply` argument; a re-arm that fails leaves the prior registration and the reply it references exactly as they were, including when the acknowledgement it owes cannot be recorded. -Posting that reply is best effort by design: the listener consumes the staged file only once its own setup and the board artifact have checked out, so the one loss window is a rare crash between that consume and the call it feeds, which drops that round's reply rather than posting it twice, and nothing here keeps a receipt, retry, or idempotency record - robust reply delivery waits on lavish-axi's exclusive listener. -The captured result is stored with immutable task-owner routing evidence and delivered directly to that task's steering inbox, without a firstmate `check` wake for the captain's words. -Filing that steering note away is not acknowledging the round, so while the round stays open every reconcile puts a live note back in the owner's inbox rather than ringing a filed one. -A task-owned source with an unhandled capture is not relaunched, so delivery failure cannot consume a round and start another poll. -That record is the only ownership evidence there is, so while any captured round of it is unacknowledged every retirement path refuses - the runner's own terminal retirement and an explicit `retire` alike - and the refusal names the acknowledgement that releases it. -A terminal result, including `session_ended`, an empty End, or missing, is delivered to the owner with an explicit stop-and-conclude instruction and is never auto-rearmed. -That round keeps the board with its owner: the source record is not retired while the terminal capture is unacknowledged, so no second armer can take the board, and acknowledging it with `bin/fm-procevent.sh handled ` is what concludes and retires it. -That conclude retains the registration it is retiring, removes it, then records the acknowledgement and restores the registration if that record cannot be written, so a failed conclude never leaves the round open with its owner gone. -An interruption between those two durable steps leaves the board unregistered with its terminal round still open, which nothing relaunches and the same `handled` call finishes. -It concludes only a round that is still open, so a repeated acknowledgement of an already-closed round reports `already-handled` and never touches whatever registration holds the board by then. -A second armer is refused with the current owner named, and the source list derives `listening`, `round-open`, or `dead` from the claim and handled captures without a second ownership record. -If the hosting worker cannot be recovered, relaunch a worker to re-host first; guarded firstmate adoption is an explicit last resort only after the old claim is proved dead. -The cross-home gap between worker rounds remains an accepted residual until lavish-axi's exclusive listener lands. -The interim crew instruction emitted by `bin/fm-brief.sh` points workers at this arm-and-acknowledge contract. - -The `when` adapter (`bin/fm-procevent-when.sh`) turns this channel into a condition->action primitive: it registers a deterministic condition and a deterministic action once, its blocking child polls the condition without waking firstmate, and a stable true fires the action at most once before one terminal outcome is durably captured and published as a wake that remains eligible for re-announcement until handled. + +**Stage an agent reply** + +Re-arm never acquires, releases, or hands off the source claim. +It may carry `--agent-reply-file `. +The file's contents are copied into that generation's private staging file and passed once to the published `--agent-reply` argument. + +A failed re-arm leaves the prior registration and its referenced reply unchanged, including when its required acknowledgement cannot be recorded. +Reply posting is best effort by design. +The listener consumes the staged file only after validating its own setup and the board artifact. +The one loss window is a rare crash between consuming the file and making the call, which drops that round's reply rather than posting it twice. + +This path keeps no receipt, retry, or idempotency record. +Robust reply delivery waits on lavish-axi's exclusive listener. + +**Deliver feedback to the worker** + +- The captured result is stored with immutable task-owner routing evidence and delivered directly to that task's steering inbox, without a firstmate `check` wake for the captain's words. +- Filing that steering note away is not acknowledging the round, so while the round stays open every reconcile puts a live note back in the owner's inbox rather than ringing a filed one. +- A task-owned source with an unhandled capture is not relaunched, so delivery failure cannot consume a round and start another poll. +- That record is the only ownership evidence there is, so while any captured round of it is unacknowledged every retirement path refuses - the runner's own terminal retirement and an explicit `retire` alike - and the refusal names the acknowledgement that releases it. + +**Conclude a terminal round** + +- A terminal result, including `session_ended`, an empty End, or missing, is delivered to the owner with an explicit stop-and-conclude instruction and is never auto-rearmed. +- That round keeps the board with its owner: the source record is not retired while the terminal capture is unacknowledged, so no second armer can take the board, and acknowledging it with `bin/fm-procevent.sh handled ` is what concludes and retires it. +- That conclude retains the registration it is retiring, removes it, then records the acknowledgement and restores the registration if that record cannot be written, so a failed conclude never leaves the round open with its owner gone. +- An interruption between those two durable steps leaves the board unregistered with its terminal round still open, which nothing relaunches and the same `handled` call finishes. +- It concludes only a round that is still open, so a repeated acknowledgement of an already-closed round reports `already-handled` and never touches whatever registration holds the board by then. + +**Ownership and recovery** + +- A second armer is refused with the current owner named, and the source list derives `listening`, `round-open`, or `dead` from the claim and handled captures without a second ownership record. +- If the hosting worker cannot be recovered, relaunch a worker to re-host first; guarded firstmate adoption is an explicit last resort only after the old claim is proved dead. +- The cross-home gap between worker rounds remains an accepted residual until lavish-axi's exclusive listener lands. +- The interim crew instruction emitted by `bin/fm-brief.sh` points workers at this arm-and-acknowledge contract. + +**Register deterministic condition and action watches** + +The `when` adapter (`bin/fm-procevent-when.sh`) registers a deterministic condition and action once. +Its blocking child polls the condition without waking firstmate. +A stable true fires the action at most once. +One terminal outcome is then durably captured and published as a wake, which remains eligible for re-announcement until handled. + The (condition, action) spec is stored privately under `state/when/` and hash-bound by a trust record the same way `bin/fm-check-register.sh` binds a custom check, while the spec separately binds the resolved action executable's bytes; a mutated or unregistered spec or a changed action executable is refused before the action runs, and that binding is reloaded from disk immediately before each fire rather than trusted from when polling started. A repo update that fast-forwards an in-repo action's bytes in place would otherwise desync every already-armed watch's trust binding with no tampering involved; `bin/fm-procevent-when.sh rebind-all` re-hashes and republishes the binding for every registered watch whose action lives under `FM_ROOT`, including one already polling, so it keeps firing across such an update instead of being refused on its next fire. + Every failure path - a mutated spec or action executable, a condition error past its budget, an expired deadline, a failed action, or an earlier fire whose outcome was never captured - produces a terminal captured outcome that wakes firstmate rather than a silent retry, and a durable single-fire marker claimed before the action makes restarts and re-polls unable to fire it twice. The adapter automates only the exact deterministic subset: anything needing judgment, and anything destructive, irreversible, or security-sensitive, keeps the ordinary check-fires-then-firstmate-decides flow, and the adapter's header and `--help` own its commands, flags, and outcome document. +**Capture and publish results** + This section is the single owner of the runner's operating contract. -Process-event commands resolve the state root to its physical directory before validating it and deriving paths, so a home reached through a symlinked ancestor behaves like its physical spelling while an unsafe target directory remains refused. -Registration writes one private record under `state/procevent/`, and a completed result plus its immutable adapter identity are captured under `state/procevent-inbox/` before any announcement or event can reference it. -By default, results are published as ordinary `check` wakes carrying the source id and committed result sequence through the existing durable wake queue, so the runner adds no second notification control plane. -The self-announcing adapter exception and its fail-safe ordering are defined below. -The watcher delivers a queued result on its ordinary cycle by reporting it as an actionable `check` wake, so a default or fallback publication reaches firstmate through the same rewake path every other wake uses and never waits for a manual drain. -A queued `check` delivery is reported at most once per captured source and sequence while any records for that key remain queued. -A durable handled acknowledgement stops future source re-announcement, while a record already queued remains under the durable queue's authority until the ordinary drain's sequence-bound post-handling acknowledgement consumes it. + +- Process-event commands resolve the state root to its physical directory before validating it and deriving paths, so a home reached through a symlinked ancestor behaves like its physical spelling while an unsafe target directory remains refused. +- Registration writes one private record under `state/procevent/`, and a completed result plus its immutable adapter identity are captured under `state/procevent-inbox/` before any announcement or event can reference it. +- By default, results are published as ordinary `check` wakes carrying the source id and committed result sequence through the existing durable wake queue, so the runner adds no second notification control plane. +- The self-announcing adapter exception and its fail-safe ordering are defined below. +- The watcher delivers a queued result on its ordinary cycle by reporting it as an actionable `check` wake, so a default or fallback publication reaches firstmate through the same rewake path every other wake uses and never waits for a manual drain. +- A queued `check` delivery is reported at most once per captured source and sequence while any records for that key remain queued. +- A durable handled acknowledgement stops future source re-announcement, while a record already queued remains under the durable queue's authority until the ordinary drain's sequence-bound post-handling acknowledgement consumes it. + +**Reconcile sources** Discovery is never a timer. -Each registered source has its own child process blocking on that source, and the watcher's per-cycle `reconcile` republishes every captured result with no durable handled acknowledgement yet - regardless of any earlier publication - restarts a source whose owner is gone, and stops this home's runner when reconciliation runs after its registration disappeared unexpectedly. +Each registered source has its own child process blocking on that source. +On every cycle, the watcher's `reconcile`: + +- Republishes every captured result without a durable handled acknowledgement, regardless of earlier publication. +- Restarts a source whose owner is gone. +- Stops this home's runner if its registration disappeared unexpectedly. + In supported steady state, a home with no registered source runs nothing, generates no state, and keeps its ordinary cadence. +**Suppress only adapter-confirmed no-op results** + Whether a captured result is a routine no-op is adapter knowledge too, and the runner names no adapter-specific condition for it either. -Before publishing, the runner asks the immutable captured owner through the built-in `silent` command or external `result.silent` operation and treats exit 0 as the only silence verdict: the result is recorded as durably handled and never announced, so it neither wakes a handler now nor returns on a later reconcile. -The task-owned terminal exception is evaluated first, so an empty terminal board round goes to its owner's steering inbox for the required conclusion instead of entering this generic silence path. -A missing command, an error, any other exit, or a silence the runner cannot durably record all publish the `check` wake exactly as before, so an adapter with no notion of a no-op needs no change and an unknown or degraded result always reaches its handler. -For built-ins, silence remains independent of the keyed-answer feed below: suppressing an announcement never suppresses the captain's own answer. -For Lavish that verdict covers two shapes - a session the adapter classifies `ended` that carries no queued content block at all, which is a review surface closed with nothing said, and `browser_disconnected` (classified `disconnected`), which carries no answer while the session remains open. -Any recognized top-level `prompts` or `feedback` block counts as content regardless of its declared count, and a malformed header makes the result indeterminate rather than empty. -A `Send & End` close carrying the captain's answer arrives as `status: feedback` with `session_ended`, so it classifies `feedback` and is announced unchanged, as is any `ended` result that still carries content, and every `waiting`, `missing`, `unknown`, or unreadable result. + +- Before publishing, the runner asks the immutable captured owner through the built-in `silent` command or external `result.silent` operation and treats exit 0 as the only silence verdict: the result is recorded as durably handled and never announced, so it neither wakes a handler now nor returns on a later reconcile. +- The task-owned terminal exception is evaluated first, so an empty terminal board round goes to its owner's steering inbox for the required conclusion instead of entering this generic silence path. +- A missing command, an error, any other exit, or a silence the runner cannot durably record all publish the `check` wake exactly as before, so an adapter with no notion of a no-op needs no change and an unknown or degraded result always reaches its handler. +- For built-ins, silence remains independent of the keyed-answer feed below: suppressing an announcement never suppresses the captain's own answer. + +**Lavish silence rules** + +- For Lavish that verdict covers two shapes - a session the adapter classifies `ended` that carries no queued content block at all, which is a review surface closed with nothing said, and `browser_disconnected` (classified `disconnected`), which carries no answer while the session remains open. +- Any recognized top-level `prompts` or `feedback` block counts as content regardless of its declared count, and a malformed header makes the result indeterminate rather than empty. +- A `Send & End` close carrying the captain's answer arrives as `status: feedback` with `session_ended`, so it classifies `feedback` and is announced unchanged, as is any `ended` result that still carries content, and every `waiting`, `missing`, `unknown`, or unreadable result. + +**Retire terminal sources** Whether a captured result ends its source is adapter knowledge, never the runner's. -After capture - and after initial `check` publication for the default ordering - the runner asks the immutable captured owner through the built-in `terminal` command or external `result.terminal` operation and retires the registration on exit 0 alone - except a task-owned board, whose terminal retirement is refused until its owner acknowledges the round, as the crew-hosted section above defines - dropping only the exact registration generation captured by its claim and releasing that claim only after removal succeeds under one source boundary; a missing command, an error, or any other exit keeps the source armed, so an adapter with no notion of ending needs no change. -A failed terminal removal stays durably terminal and is completed by ordinary reconciliation without restarting its poll, while a concurrently replaced registration survives and becomes independently runnable after the old claim releases. -Any registration refuses to replace an external registration while its prior runner claim is live, uncertain, orphaned, or terminal-pending; replacement becomes eligible only after that generation is proved gone or its terminal retirement completes. -A source that has ended therefore captures at most one terminal result, is never restarted, and leaves no recurring poll work. -For ordinary sources, explicit `retire` stays the supported and idempotent path afterwards; a task-owned board instead refuses `retire` until its owner concludes the open terminal round with `handled`. -For Lavish that verdict covers an ended session, a missing session, and the final feedback of a `Send & End` review, which the published poll marks with `session_ended` before it returns only empty ended sessions. +After capture, the runner asks the immutable captured owner whether the result is terminal. +It uses the built-in `terminal` command or external `result.terminal` operation. +Under the default ordering, this happens after the initial `check` publication. + +- Exit 0 retires the registration. + The exception is a task-owned board, whose owner must first acknowledge the round as defined above. +- Retirement drops only the exact registration generation captured by the claim. + Under one source boundary, it releases that claim only after removal succeeds. +- A missing command, an error, or any other exit keeps the source armed. + An adapter with no notion of ending needs no change. + +- A failed terminal removal stays durably terminal and is completed by ordinary reconciliation without restarting its poll, while a concurrently replaced registration survives and becomes independently runnable after the old claim releases. +- Any registration refuses to replace an external registration while its prior runner claim is live, uncertain, orphaned, or terminal-pending; replacement becomes eligible only after that generation is proved gone or its terminal retirement completes. +- A source that has ended therefore captures at most one terminal result, is never restarted, and leaves no recurring poll work. +- For ordinary sources, explicit `retire` stays the supported and idempotent path afterwards; a task-owned board instead refuses `retire` until its owner concludes the open terminal round with `handled`. +- For Lavish that verdict covers an ended session, a missing session, and the final feedback of a `Send & End` review, which the published poll marks with `session_ended` before it returns only empty ended sessions. + +**Apply built-in results automatically** Applying a captured result through code is a built-in adapter seam, and some built-in results carry no judgement at all: they must simply be applied idempotently to this home's own durable state. Leaving that to a handler means it can silently not happen, so immediately after the terminal check above the runner calls `bin/fm-procevent-.sh autohandle ` and lets the built-in adapter apply and acknowledge its own result. + That call runs strictly after terminal retirement, because a handling adapter re-arms its own next source and retiring afterwards would drop that fresh registration and leave the source silently dead. Exit 0 means the adapter fully applied and acknowledged the result; a missing command, an error, or any other exit is not a capture failure but leaves the result unacknowledged and therefore still eligible for re-announcement, so a handler receives it exactly as before and an adapter with no such command needs no change. -Announcement ordering is adapter-declared through `bin/fm-procevent-.sh self-announcing`: an adapter that answers exit 0 declares that every result its autohandle fully applies is announced through a durable downstream channel of its own, so the runner applies first and publishes a `check` wake only for what remains unhandled afterwards; every other adapter keeps the strict publish-before-apply order, and its autohandle runs only when this capture's own wake was successfully appended to the durable queue. + +**Adapter-controlled announcement order** + +The built-in `bin/fm-procevent-.sh self-announcing` command declares announcement order: + +| Response | Runner behavior | +| --- | --- | +| Exit 0 | The adapter declares that every result its autohandle fully applies is announced through its own durable downstream channel; the runner applies first, then publishes a `check` wake only for results still unhandled. | +| Any other response | Keep strict publish-before-apply ordering; autohandle runs only after this capture's own wake was successfully appended to the durable queue. | + The remote-secondmate reply adapter declares itself self-announcing: a captured reply reaches its local status mirror and settles its correlated pending-reply expectation without any handler step, the mirrored status bytes are the single wake for one remote note through the same signal classification a local secondmate's append gets, and only a capture the adapter could not fully apply is published as a `check` wake, whose adapter handling remains idempotent. The [remote-secondmate channel contract](remote-secondmates.md#normal-operation) owns replay suppression and its bounded upgrade exception; a replay that adds no mirror bytes stays quiet. +**Feed keyed captain answers** + Keyed captain answers from built-in adapters use one more seam of the same kind, and the runner still decides nothing about them. Some built-in sources carry the captain's answer to a captain-held task, and what such an answer means is owned once by `bin/fm-captain-hold.sh`'s keyed-answer intake rather than by any channel. -A built-in source bound with `bin/fm-captain-hold.sh bind` therefore has each captured result passed to `bin/fm-procevent-.sh answers `, and whatever that prints is piped straight into that intake. -A binding can select one decision origin or the script's cross-origin mode; the command header owns the exact forms and key interpretation. -The built-in adapter reports only what the captain chose; the intake owns every rule about what happens next, so the runner names no adapter, parses no result, and carries no decision rule, and a future built-in answer source needs nothing here beyond an `answers` command and a binding. -The reserved Reconcile selection uses the parallel optional `reconciles` adapter command and binding-verified `reconcile-requests` intake rather than entering keyed answers; [`captain-hold-lifecycle.md`](captain-hold-lifecycle.md#reconcile-re-check-reality-never-a-blind-close) owns those semantics. -Feeding is independent of handling: it never acknowledges a result and never suppresses a wake, because recording the answer or request is transcription while acting on it is firstmate's judgement. -An unbound built-in source, a built-in adapter without the corresponding command, and a failure on either side all leave the capture untouched and still announced. -External binding responses never enter either authority-bearing intake. + +- A built-in source bound with `bin/fm-captain-hold.sh bind` therefore has each captured result passed to `bin/fm-procevent-.sh answers `, and whatever that prints is piped straight into that intake. +- A binding can select one decision origin or the script's cross-origin mode; the command header owns the exact forms and key interpretation. +- The built-in adapter reports only what the captain chose; the intake owns every rule about what happens next, so the runner names no adapter, parses no result, and carries no decision rule, and a future built-in answer source needs nothing here beyond an `answers` command and a binding. + +**Reconcile selections and handling boundaries** + +- The reserved Reconcile selection uses the parallel optional `reconciles` adapter command and binding-verified `reconcile-requests` intake rather than entering keyed answers; [`captain-hold-lifecycle.md`](captain-hold-lifecycle.md#reconcile-re-check-reality-never-a-blind-close) owns those semantics. +- Feeding is independent of handling: it never acknowledges a result and never suppresses a wake, because recording the answer or request is transcription while acting on it is firstmate's judgement. +- An unbound built-in source, a built-in adapter without the corresponding command, and a failure on either side all leave the capture untouched and still announced. +- External binding responses never enter either authority-bearing intake. + +**Machine-wide source ownership** Ownership is machine-wide per canonical source, because separate homes can share one underlying source store. -Claims live under `$XDG_STATE_HOME/firstmate/procevent-claims` (override with `FM_PROCEVENT_CLAIM_ROOT`). -Each claim binds its caller-reported home and runner PID to a process identity, unique claim generation, exact registration-file generation, and resolved state-root identity. -Registration, acquisition, replacement, retirement, and generation-bound release are serialized at one machine-wide boundary per source. -A live identity-matched owner is never displaced, and release removes only the exact generation the caller acquired. -Every stop proves ownership before its first signal: the live runner's recorded process identity must match and it must still lead its process group. -Once that stop has proved ownership and sent TERM, its own escalation to KILL checks only whether the proved group still has members; it does not re-read the leader's identity or group membership, which can change or become unreadable as TERM ends the leader. -This proof belongs only to that stop's own escalation and cannot authorize another caller that encounters an unproved group. + +- Claims live under `$XDG_STATE_HOME/firstmate/procevent-claims` (override with `FM_PROCEVENT_CLAIM_ROOT`). +- Each claim binds its caller-reported home and runner PID to a process identity, unique claim generation, exact registration-file generation, and resolved state-root identity. +- Registration, acquisition, replacement, retirement, and generation-bound release are serialized at one machine-wide boundary per source. +- A live identity-matched owner is never displaced, and release removes only the exact generation the caller acquired. + +**Prove ownership before stopping a runner** + +- Every stop proves ownership before its first signal: the live runner's recorded process identity must match and it must still lead its process group. +- Once that stop has proved ownership and sent TERM, its own escalation to KILL checks only whether the proved group still has members; it does not re-read the leader's identity or group membership, which can change or become unreadable as TERM ends the leader. +- This proof belongs only to that stop's own escalation and cannot authorize another caller that encounters an unproved group. + +**Recover orphaned claims** + A stale claim whose process group still has members is one `reconcile` never displaces, and the two shapes it comes in recover differently. `reconcile` preserves such a claim without signalling the ambiguous group or starting a replacement: the group check probes the runner's own process group, which contains its polling source child, so surviving members can mean that child is still attached to the session the source collects from, and a replacement would put a second destructive poller on it. + `list` reports both shapes as `orphaned`. -When the recorded pid is alive under a different identity while the group still has members, the claim boundary itself does not consult the process group, so `bin/fm-procevent.sh start ` reclaims that claim provided the dead generation's reservation records can still be tidied, and otherwise refuses with `cannot claim source`; that tidy-up is waived only for a generation proven gone, which this one is not. + +**Reused PID with surviving group members** + +When the recorded pid is alive under a different identity but the group still has members, the claim boundary itself does not consult the process group. +In this case, `bin/fm-procevent.sh start ` reclaims the claim only if it can tidy the dead generation's reservation records. +Otherwise it refuses with `cannot claim source`. + +Tidy-up is waived only for a generation proved gone. +This generation does not meet that condition. That hand-run command is the recovery path, taken by someone who has checked that nothing is still polling the source. + That asymmetry between the automatic path and the deliberate one is the design rather than an inconsistency, and it is not a claim-level invariant: nothing below `reconcile` enforces it. -When the leader itself is gone and its group still has members - the leader died to anything other than the stop's own signal - `start` does not reclaim the claim either: it reports `already owned` and changes nothing, and `retire`, `reconcile`, `sweep-home`, and the guard all refuse the surviving group permanently, so the source stops listening. + +**Dead leader with surviving group members** + +If the leader died from anything other than the stop's own signal and its group still has members, `start` does not reclaim the claim. +It reports `already owned` and changes nothing. + +`retire`, `reconcile`, `sweep-home`, and the guard all refuse the surviving group permanently, so the source stops listening. Recovery there is a human verifying whether the dead runner's polling child is still attached to the source; once that process group is empty the generation reads as gone and the next `reconcile` reclaims the source on its own. + Nothing automatic signals that group, and whether it may ever be signalled remains an open decision; the repaired guard does not close this gap. -Neither shape stops listening quietly: the first `reconcile` that strands a claim generation publishes a durable `check` wake naming the source and what clears it - the `start` command for the reused pid, the check to make for the leaderless group - and later cycles stay silent for that same generation while a genuinely new stranded claim announces again. + +**Report stranded claims** + +The first `reconcile` that strands either kind of claim generation publishes a durable `check` wake. +It names the source and the recovery step: + +- For a reused pid, the `start` command. +- For a leaderless group, the check to make. + +Later cycles stay silent for the same generation. +A genuinely new stranded claim announces again. + +**Reclaim a generation proved gone** + Reclaiming a generation that IS gone is not gated on tidying anything that generation left behind: its capture-reservation records, its staging file, or the registry directory a claim recorded for them. Every one of those is keyed by claim token and every replacement claims a fresh one, so a leftover that can no longer be located or removed - a state-root identity a claim recorded before its home was re-created, or a recorded registry directory that no longer resolves to a directory - is stale bytes rather than an ownership hazard. + Making any of them a precondition is what leaves a provably dead runner owning its source permanently, because none of those conditions clears on its own. -Ordinary release and reclamation still attempt reservation cleanup and require it unless both owner staleness and whole-group absence prove the generation gone. -The narrow live-owner terminal-self-retirement path also attempts cleanup but tolerates its own still-in-flight reservation, which the runner removes on the normal end-of-capture path; exact home, PID, and claim-token ownership remains mandatory before the claim is released. -If identity cannot be established before the first signal, or a surviving owned group cannot be proved stopped, the operation preserves the registration and claim for safe retry rather than adding a second owner. -A live PID whose identity no longer matches is refused before the first signal. -Identity and process-group verification cannot be made atomic with signalling in portable shell: the reaper signals only a target it has verified as the recorded generation, but PID and group reuse remain possible in the narrow interval between verification and the signal. -Launch pacing is the primary host-wedge protection; watchdog cleanup is a backstop. - -Supported secondmate retirement preflights each target home's bounded `sweep-home` command before destructive teardown, snapshots its registrations outside the target, then runs the sweep at that home's final deletion or return boundary. -If deletion or return fails, teardown restores those registrations and reconciles them before returning the refusal. -If restoration or rearming also fails, teardown returns a distinct status and reports the retained registration backup path for manual recovery instead of hiding the retired waits. -The sweep retires local registrations and machine-wide claims whose recorded state-root identity matches that home's resolved state root through the same identity-checked, generation-bound retirement path, and leaves foreign-home claims untouched. -Teardown refuses with the home, lease, routing evidence, registrations, claims, and runners retained when identity is uncertain, ownership is unreadable or unreleased, or relevant state exists without a sweep-capable child script. + +- Ordinary release and reclamation still attempt reservation cleanup and require it unless both owner staleness and whole-group absence prove the generation gone. +- The narrow live-owner terminal-self-retirement path also attempts cleanup but tolerates its own still-in-flight reservation, which the runner removes on the normal end-of-capture path; exact home, PID, and claim-token ownership remains mandatory before the claim is released. + +**Stop refusals and residual races** + +- If identity cannot be established before the first signal, or a surviving owned group cannot be proved stopped, the operation preserves the registration and claim for safe retry rather than adding a second owner. +- A live PID whose identity no longer matches is refused before the first signal. +- Identity and process-group verification cannot be made atomic with signalling in portable shell: the reaper signals only a target it has verified as the recorded generation, but PID and group reuse remain possible in the narrow interval between verification and the signal. +- Launch pacing is the primary host-wedge protection; watchdog cleanup is a backstop. + +**Retire a secondmate home** + +- Supported secondmate retirement preflights each target home's bounded `sweep-home` command before destructive teardown, snapshots its registrations outside the target, then runs the sweep at that home's final deletion or return boundary. +- If deletion or return fails, teardown restores those registrations and reconciles them before returning the refusal. +- If restoration or rearming also fails, teardown returns a distinct status and reports the retained registration backup path for manual recovery instead of hiding the retired waits. +- The sweep retires local registrations and machine-wide claims whose recorded state-root identity matches that home's resolved state root through the same identity-checked, generation-bound retirement path, and leaves foreign-home claims untouched. +- Teardown refuses with the home, lease, routing evidence, registrations, claims, and runners retained when identity is uncertain, ownership is unreadable or unreleased, or relevant state exists without a sweep-capable child script. + +**Recover from unsupported manual deletion** + Raw manual deletion of a Firstmate home is unsupported because it can orphan a blocking child. To recover, restore that home's tracked `bin/fm-procevent.sh`, run `FM_HOME= /bin/fm-procevent.sh sweep-home`, then rerun the supported teardown. + The owning-home lease below bounds how long such an orphan can run, but it is a backstop, not a substitute for the supported path. +**Home lease and its limits** + A runner is bound to the HOME that owns it, not to the one session that armed it. That granularity is deliberate: a persistent source is meant to outlive the turn and the session that armed it, so binding a runner to its arming session would stop exactly the sources this mechanism exists to keep running. + Any activity in the same home refreshes the lease, so a replacement session, another watcher, or an ordinary inspection command keeps a runner of that home alive; a runner whose SOURCE is no longer wanted in a live home is stopped by reconcile when that source is retired, independently of the lease. The lease is therefore the backstop for a home that is GONE - the torn-down test sandbox this change exists to bound - and not a per-session ownership check. + KNOWN LIMIT: while any activity continues in a home whose original owning session has ended, that activity refreshes the lease and a runner of that home keeps running until its source is retired or the home goes away. + +**Keep and guard the lease** + Detaching a runner into its own process group is what lets a persistent source outlive the turn that armed it, and on its own it is also what lets a runner outlive its whole home: reparented to init, it keeps its blocking child - and every process that child spawns - running with nothing left to reap it. -So a home's process-event state carries a lease that registration, attached start, reconciliation, acknowledgement, and listing refresh, and the watcher's reconcile cycle is what keeps it fresh in a live home. -An attached public `start` continues refreshing the lease while its caller remains attached. -Each runner fails closed unless a small guard starts successfully beside it in a separate process group. -That guard accepts the lease only while the state root retains the device/inode identity recorded by the runner's claim, and initiates the verified stop after two consecutive reads cannot prove that identity and lease freshness, so one unreadable read cannot kill a live runner. -Those two reads are spaced half a check interval apart, so the pair the debounce requires completes inside one check interval instead of costing two of them. -For a runner whose ownership can still be proved, the nominal detection bound is therefore the lease plus one check interval, after which the verified stop runs within its own grace period; the lease age is compared in whole seconds, so a configured lease is honoured until that age reads one second past it, and scheduling delays or failed inspection and signalling can extend the whole bound. + +- So a home's process-event state carries a lease that registration, attached start, reconciliation, acknowledgement, and listing refresh, and the watcher's reconcile cycle is what keeps it fresh in a live home. +- An attached public `start` continues refreshing the lease while its caller remains attached. +- Each runner fails closed unless a small guard starts successfully beside it in a separate process group. +- That guard accepts the lease only while the state root retains the device/inode identity recorded by the runner's claim, and initiates the verified stop after two consecutive reads cannot prove that identity and lease freshness, so one unreadable read cannot kill a live runner. +- Those two reads are spaced half a check interval apart, so the pair the debounce requires completes inside one check interval instead of costing two of them. + +**Detection and stop timing** + +For a runner whose ownership can still be proved, the nominal detection bound is the lease plus one check interval. +The verified stop then runs within its own grace period. +The lease age is compared in whole seconds, so the configured lease is honoured until that age reads one second past it. + +Scheduling delays or failed inspection and signalling can extend the whole bound. That grace is a ceiling rather than a delay every stop pays: two seconds for the ordinary signal and two more for the forced one, spent only by a group that outlives the signal it was sent, which is why a healthy runner's stop completes in a fraction of a second. + The group signal reaches the blocking child and everything under it exactly as retirement does. -A runner exports the inherited `FM_PROCEVENT_IN_RUNNER` marker and every lease refresh is skipped under it, so a runner and its ordinary children do not certify their own owner, and the next reconcile in a live home simply starts a replacement runner. -That no-self-refresh rule is CONFUSED-AGENT-GRADE, the same deliberate captain-decided grade `bin/fm-lease-lib.sh` documents: it stops the accidental case this boundary exists for, an orphaned or test-scaffolding source tree that would otherwise keep its own owner alive. -A source that DELIBERATELY strips the marker from its environment can still refresh the lease, so adversarial-grade unforgeability is explicitly out of scope here and tracked as separate follow-up design work. -Scope is the owning state root and one runner generation, never a script or process name, so a live source in another home is untouched: that home refreshes its own lease. -`FM_PROCEVENT_OWNER_LEASE_SECONDS` (default 600, range 1..86400) is how long a runner keeps going with no sign of activity in its owning home, and `FM_PROCEVENT_OWNER_CHECK_SECONDS` (default 15, range 1..3600) is the guard's detection interval: it re-reads the lease and the recorded state-root identity twice within each interval, half an interval apart, so the two reads its debounce needs fit inside one interval rather than costing two. -`FM_PROCEVENT_LAUNCH_FLOOR_SECONDS` (default 1, range 1..3600) is the minimum time between consecutive launches of one registration generation's stored command, bounding the launch rate of an immediately returning source during that lease window. + +**Prevent accidental self-refresh** + +- A runner exports the inherited `FM_PROCEVENT_IN_RUNNER` marker and every lease refresh is skipped under it, so a runner and its ordinary children do not certify their own owner, and the next reconcile in a live home simply starts a replacement runner. +- That no-self-refresh rule is CONFUSED-AGENT-GRADE, the same deliberate captain-decided grade `bin/fm-lease-lib.sh` documents: it stops the accidental case this boundary exists for, an orphaned or test-scaffolding source tree that would otherwise keep its own owner alive. +- A source that DELIBERATELY strips the marker from its environment can still refresh the lease, so adversarial-grade unforgeability is explicitly out of scope here and tracked as separate follow-up design work. +- Scope is the owning state root and one runner generation, never a script or process name, so a live source in another home is untouched: that home refreshes its own lease. + +**Lease and launch pacing settings** + +| Setting | Default | Range | Purpose | +| --- | --- | --- | --- | +| `FM_PROCEVENT_OWNER_LEASE_SECONDS` | 600 | 1..86400 | How long a runner continues without activity in its owning home. | +| `FM_PROCEVENT_OWNER_CHECK_SECONDS` | 15 | 1..3600 | Guard detection interval; it reads the lease and recorded state-root identity twice per interval, half an interval apart, so both debounce reads fit inside one interval. | +| `FM_PROCEVENT_LAUNCH_FLOOR_SECONDS` | 1 | 1..3600 | Minimum time between consecutive launches of one registration generation's stored command; bounds immediately returning sources during the lease window. | + The generation's first launch is immediate, later launches share its monotonic pacing timestamp, a timestamp from before a reboot is treated as expired, and replacing the registration starts a fresh pacing generation. +**Confirm detached launches** + `FM_PROCEVENT_LAUNCH_CONFIRM_SECONDS` (default 3, range 1..600) bounds how long `reconcile` waits for the runners it just started to prove they are running: never less than the configured value, and at most one second more, because the wait is measured on a whole-second clock. -Starting a runner is detached and its errors are not visible to the caller, so `reconcile` reports a start only after the source is observed owned or its launch-pacing stamp has advanced or appeared, and reports every unconfirmed launch as `failed=` and a non-zero exit instead. -Both signals are durable evidence a runner claimed: ownership is the only evidence a runner still blocked on its source ever shows, and the stamp - written after the claim and before the source command runs, and removed only by registration replacement - covers a runner that claimed, ran and exited between two polls. -A healthy launch therefore confirms on the first poll and the window only bounds a launch that has not yet proved itself - one that died before claiming, or one merely too slow to claim inside the window; confirmation cannot tell those apart, and a launch that proves itself on a later cycle closes its failure episode without a retraction wake. -All of a cycle's launches share one window, so a home full of sources that cannot start costs the same bounded wait as one. + +- Starting a runner is detached and its errors are not visible to the caller, so `reconcile` reports a start only after the source is observed owned or its launch-pacing stamp has advanced or appeared, and reports every unconfirmed launch as `failed=` and a non-zero exit instead. +- Both signals are durable evidence a runner claimed: ownership is the only evidence a runner still blocked on its source ever shows, and the stamp - written after the claim and before the source command runs, and removed only by registration replacement - covers a runner that claimed, ran and exited between two polls. +- A healthy launch therefore confirms on the first poll and the window only bounds a launch that has not yet proved itself - one that died before claiming, or one merely too slow to claim inside the window; confirmation cannot tell those apart, and a launch that proves itself on a later cycle closes its failure episode without a retraction wake. +- All of a cycle's launches share one window, so a home full of sources that cannot start costs the same bounded wait as one. + +**Keep confirmation below the watcher interval** Keep this window well below `FM_POLL`. `bin/fm-watch.sh` runs `reconcile` once per supervision cycle, so a source that cannot start makes every cycle wait up to the confirm window before the rest of that cycle runs. + Raising the confirm window lengthens every supervision cycle and delays wake delivery by up to that much. +**Report launch failures** + A source that can never start is reported as `failed=` with a non-zero exit on every `reconcile`, rather than counted as `started` and retried silently as though it were healthy, so a wedged source stays visible instead of presenting as armed. -That count reaches only whoever runs the command, because `bin/fm-watch.sh` discards `reconcile`'s output and exit status, so an unconfirmed launch is also announced through the wake queue: `reconcile` publishes a durable `check` wake (`procevent::launch-failed:-`) once per failure episode, and later cycles stay silent for that episode until a launch of that source confirms, after which a fresh failure announces again under a fresh key, because the watcher never re-surfaces a key it has already surfaced. -The announcement changes nothing about the launch: `reconcile` keeps relaunching the source every cycle exactly as before, and nothing is retried differently, throttled, or recovered from that signal. -The wake says only what was observed for that shape - the launch did not prove it took the claim within the window - and, if it stays that way, names the source command and adapter binary the registration names as what to check and the attached `bin/fm-procevent.sh start ` as what reproduces a refusal on stderr, where the detached launch discards it; a later cycle that finds the source owned ends the episode on its own, so a runner that was merely slow to claim needs nothing from the operator. -A source stranded on a claim nothing may automatically displace is announced the same way, once per stranded claim generation, as described above. -`bin/fm-watch.sh` surfaces both under their own headlines - `process-event source stranded` and `process-event source failed to start` - rather than as a captured result. +The `failed=` count reaches only the command's caller because `bin/fm-watch.sh` discards `reconcile` output and exit status. +For that reason, `reconcile` also publishes a durable `check` wake once per failure episode, with key `procevent::launch-failed:-`. +Later cycles stay silent for that episode until a launch confirms. +A later fresh failure gets a fresh key, because the watcher never re-surfaces a key it has already surfaced. + +- The announcement changes nothing about the launch: `reconcile` keeps relaunching the source every cycle exactly as before, and nothing is retried differently, throttled, or recovered from that signal. +- The wake reports only the observed failure: the launch did not prove that it took the claim within the window. +- If the failure persists, inspect the source command and adapter binary named in the registration. + The wake names both, along with the attached `bin/fm-procevent.sh start ` command that reproduces the refusal on stderr. + The detached launch discards that output. +- A later cycle that finds the source owned ends the episode automatically. + A runner that was merely slow to claim needs no operator action. +- A source stranded on a claim nothing may automatically displace is announced the same way, once per stranded claim generation, as described above. +- `bin/fm-watch.sh` surfaces both under their own headlines - `process-event source stranded` and `process-event source failed to start` - rather than as a captured result. + +**Reject unusable settings** A value this command cannot use is refused by name before anything is launched, the same way `FM_PROCEVENT_LAUNCH_FLOOR_SECONDS` and `FM_PROCEVENT_MAX_OUTPUT_BYTES` are refused, so a mistyped window can never present as a fleet of sources that cannot start. `bin/fm-watch.sh` validates the same value when it arms and refuses to arm on an unusable one, naming the variable and the range: under a running watcher that refusal would otherwise repeat on every cycle into a discarded stdout and leave the whole home disarmed while presenting as supervised, whereas a watcher that will not arm is loud through the liveness guard. +**Limit captured output** + `FM_PROCEVENT_MAX_OUTPUT_BYTES` (default 1048576) bounds a single captured result while the source runs; oversized output is drained but truncated with a stderr notice rather than staged or published whole or dropped. +**Durability guarantees and limits** + The runner proves exactly one durability boundary: output that reached the runner is stored at mode `0600` before any event referencing it is published, and a captured result with no durable handled acknowledgement remains eligible for bounded re-announcement across any number of drains and restarts, not only the crash window right after capture. -`bin/fm-procevent.sh handled ` is the only thing that stops re-announcement: a generation-keyed, private, path-safe, durable, and idempotent acknowledgement that atomically checks and deduplicates by the exact source and sequence, so a paired effect gated on its first-time-vs-repeat report is never authorized twice. -Default and fallback `check` publication is still best-effort, so the same source and sequence can repeat even before any restart; handlers deduplicate that identity rather than assuming a wake is unique. -The runner proves nothing about the source side, and the handled acknowledgement proves nothing about a paired external effect performed before it: a crash between that effect and the acknowledgement call can still repeat the effect on replay, so this is never a generic exactly-once guarantee. -The published `lavish-axi poll` clears feedback destructively before returning it, so a result lost between that clearing and the runner reading process output is unrecoverable. -Never describe this path as at-least-once, no-loss, or lossless. + +- `bin/fm-procevent.sh handled ` is the only thing that stops re-announcement: a generation-keyed, private, path-safe, durable, and idempotent acknowledgement that atomically checks and deduplicates by the exact source and sequence, so a paired effect gated on its first-time-vs-repeat report is never authorized twice. +- Default and fallback `check` publication is still best-effort, so the same source and sequence can repeat even before any restart; handlers deduplicate that identity rather than assuming a wake is unique. +- The runner proves nothing about the source side, and the handled acknowledgement proves nothing about a paired external effect performed before it: a crash between that effect and the acknowledgement call can still repeat the effect on replay, so this is never a generic exactly-once guarantee. +- The published `lavish-axi poll` clears feedback destructively before returning it, so a result lost between that clearing and the runner reading process output is unrecoverable. +- Never describe this path as at-least-once, no-loss, or lossless. + `docs/verification/process-event-sources.md` holds the measurements and `.agents/skills/process-event-sources/SKILL.md` owns the handling procedure. ## Spoken interface and captain inbox (config/voice-*, config/inbox-*) The spoken interface in [`docs/voice-relay.md`](voice-relay.md) and the model-backed subcommands of `bin/fm-inbox.sh` reach a paid API in a named account, so no region, model id or AWS profile is shipped as a tracked default. Each is one line in a local, gitignored `config/` file, with an environment variable that overrides it for a single run, and a missing required value refuses with the path to write rather than falling back to a value that belongs to another home. + That configuration is the whole opt-in: an unconfigured home cannot start the relay and cannot run `fm-inbox.sh say` or `ask`, while `note`, `announce`, `reply`, `receipts`, `ready`, `status`, `list` and `drain` need no configuration at all because they make no model call. The voice handover depends on `note`, so it keeps working in a home that has configured nothing. @@ -1157,8 +2181,15 @@ The voice handover depends on `note`, so it keeps working in a home that has con | `config/inbox-ask-model` | `FM_INBOX_ASK_MODEL` | Side-question model id, required by `fm-inbox.sh ask`. | | `config/inbox-profile` | `FM_INBOX_PROFILE` | AWS profile for those two calls; absent, or an explicitly empty variable, means whatever credentials are already in the environment. | +**How configuration files are parsed** + Each account, model and voice file above is read as its first line that is not blank and not a `#` comment, so a comment above the value is fine. -The two read files are parsed differently: `config/voice-read-scope` must hold the bare word and nothing but blank space around it, so a comment header there refuses instead of being skipped, while every line of `config/voice-read-deny` that is not blank and not a `#` comment is one more substring. +The two read files use different parsing rules: + +- `config/voice-read-scope` must contain only the bare word with optional blank space around it. + A comment header causes a refusal rather than being skipped. +- In `config/voice-read-deny`, every line that is neither blank nor a `#` comment adds one substring. + `FM_VOICE_RELAY` and `FM_VOICE_PYTHON` belong to the laptop rather than to a home, so they have no config file: `bin/fm-voice-client.py` requires the relay path as a flag or that variable and carries no default path. ## Environment variables @@ -1340,13 +2371,17 @@ FM_INBOX_PROFILE= # overrides config/inbox-profile; explicitly empty force `fm-teardown.sh` retries only Git's `Unable to create '...index.lock': File exists` return failure up to `FM_TREEHOUSE_RETURN_LOCK_RETRIES` times. `FM_TREEHOUSE_RETURN_LOCK_RETRIES` accepts a nonnegative integer, and an unset, blank, or invalid value uses the default of 3. + `FM_TREEHOUSE_RETURN_LOCK_RETRY_WAIT_SECS` accepts nonnegative whole or fractional seconds between attempts. When it is unset or blank, `FM_STALE_WORKTREE_LOCK_RETRY_WAIT_SECS` remains a compatible fallback, and a blank fallback uses the 1-second default. + An invalid nonblank wait falls back to 1 second rather than interrupting teardown. Teardown never removes a lock during the retry window, and after that window it attempts stale-lock cleanup only for a still-present lock that passes the configured age and live-holder checks. `fm-fleet-sync.sh` applies the same shape to an orphaned `.git/packed-refs.lock`: it retries only Git's `Unable to create '...packed-refs.lock': File exists` fetch failure up to `FM_FLEET_SYNC_PACKED_REFS_LOCK_RETRIES` times (nonnegative integer; unset, blank, or invalid uses the default of 3), waiting `FM_FLEET_SYNC_PACKED_REFS_LOCK_RETRY_WAIT_SECS` seconds (nonnegative whole or fractional; invalid falls back to 1 second) before each. Only after those retries exhaust does it remove the lock, and only when it is provably stale - still present, mtime age at least `FM_FLEET_SYNC_PACKED_REFS_LOCK_AGE_SECS` (default 30), and no `lsof` holder of the lock file or of the clone worktree itself (a live `git` keeps that as its cwd even in the window after it closes the lock and before it exits). + A live lock, a missing `lsof`, any failed check, or any other fetch failure keeps today's behavior. Every wait, retry, and removal is printed to stderr, and a successful recovery also prints one `recovered:` summary line to stdout so a session-start refresh - which discards fleet-sync stderr and relays only stdout - still surfaces it. + The shared staleness proof lives in `bin/fm-lock-lib.sh`, which both `fm-teardown.sh` and `fm-fleet-sync.sh` use. From b52d401878ada3053231dfb037715d6284b20863 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Thu, 24 Sep 2026 23:04:32 -0700 Subject: [PATCH 06/84] fix: limit project memory edits to factual corrections (#5636) * fix: bound worker edits of project AGENTS.md/CLAUDE.md to factual corrections These files are loaded into every agent session of a project, so additions should be a deliberate human choice rather than automated task output. The ship brief's project-memory section and AGENTS.md section 6 previously invited workers to record durable knowledge, which let project AGENTS.md files accrete detail the codebase or README already carries. Workers now edit only to fix factually wrong content - including content their own change made wrong - and fm-ensure-agents-md.sh runs only alongside such a correction. Stow no longer routes project-memory additions through ship tasks, and the generated skeleton no longer invites discovery-driven additions. * no-mistakes(review): Stop running fm-ensure-agents-md.sh on memory-file corrections * no-mistakes(document): Clarify manual project-memory initialization and remove duplicate guidance --- .agents/skills/stow/SKILL.md | 10 +++++----- AGENTS.md | 5 +++-- bin/fm-brief.sh | 18 ++++++++---------- bin/fm-ensure-agents-md.sh | 8 +++++--- docs/architecture.md | 9 +++------ docs/scripts.md | 2 +- tests/fm-brief.test.sh | 24 +++++++++++++++++------- 7 files changed, 42 insertions(+), 34 deletions(-) diff --git a/.agents/skills/stow/SKILL.md b/.agents/skills/stow/SKILL.md index 8b86468011d..ba5099df18c 100644 --- a/.agents/skills/stow/SKILL.md +++ b/.agents/skills/stow/SKILL.md @@ -180,8 +180,8 @@ Approved project-level destinations are not produced by stow: they ship normally Because this destination is local and untracked, it is also the JIT home for private conditional knowledge that no committed surface may hold. - An already-existing user-owned local on-demand note with an established trigger, after confirming it is untracked, private, and able to hold the quoted entry. The pass may add the entry to that existing owner but never creates a new note, skill, or trigger for this purpose. -- A project's existing committed `AGENTS.md`, for project-intrinsic knowledge useful to nearly every session of that project, through a normal crewmate ship task using `bin/fm-ensure-agents-md.sh` and the project's registered delivery mode. -- A project-level skill in the project's own repository, for situation-conditional knowledge within one project, through the same ship-task path. +- A project-level skill in the project's own repository, for situation-conditional knowledge within one project, through a normal ship task and the project's registered delivery mode. + A project's committed `AGENTS.md` is never an offload destination: crewmates correct it but only humans extend it (AGENTS.md section 6). Forbidden destinations: any firstmate-repo-tracked skill per the hard rule; firstmate's own `AGENTS.md`, which is always-loaded for every fleet session; `docs/` alone, which is never agent-loaded on demand, though a skill body may point into docs for depth; and any committed surface for private content. A local skill exists only in this home, so offloading an entry out of `data/captain-shared.md` removes it from every inheriting home's always-injected memory: the proposal must say so, and the default for shared entries is keep. @@ -190,7 +190,7 @@ A local skill exists only in this home, so offloading an entry out of `data/capt 1. Reduce non-pinned material now. For each eligible non-pinned candidate, record its first line, source file, estimated tokens, one-line trigger, live destination, privacy and visibility verdict, and actual budget relief in the completion receipt. - Autonomously relocate it only by adding it to an already-existing allowed JIT note, or by routing it through a project's established delivery path to its existing owning `AGENTS.md`, then confirming that destination holds the quoted entry before removing the memory entry. + Autonomously relocate it only by adding it to an already-existing allowed JIT note, or by routing it through a project's established delivery path to an already-existing allowed project-level destination, then confirming that destination holds the quoted entry before removing the memory entry. A destination that needs creation, uncompleted project delivery, or any other future work is not live and cannot count as relief, so continue with the next archival or eviction rung instead of leaving an over-budget proposal pending. 2. Propose pinned relocation only. For a pinned candidate, append a `proposed-offload` section with the same fields to the completion receipt, create or refresh one durable backlog item with `bin/fm-tasks-axi.sh add`, `bin/fm-tasks-axi.sh show --full`, and `bin/fm-tasks-axi.sh update --body-file ` as appropriate, then hold it through `bin/fm-captain-hold.sh hold`. @@ -220,8 +220,8 @@ A local skill exists only in this home, so offloading an entry out of `data/capt Create `data/learnings.md` only for a genuinely new local learning with no stronger owner. - In a primary home, curate shared captain preferences only under the existing primary-authoritative shared-preference contract. In a secondmate home, route a newly discovered shared preference to the main firstmate through marked status or a document pointer instead of editing the inherited file. - - Project-intrinsic knowledge never goes directly into a project's `AGENTS.md`. - Route it through a normal ship task so a crewmate records it with `bin/fm-ensure-agents-md.sh` and the project's delivery path. + - Project-intrinsic knowledge never goes into a project's `AGENTS.md` through this fleet: a crewmate edits those files only to correct factually wrong information (AGENTS.md section 6), so no ship task carries an addition. + Keep the candidate in `data/learnings.md` or surface it in the completion receipt so the captain can extend the file by hand. - Knowledge general to every Firstmate user belongs in this repo's shared tracked material through the normal branch, no-mistakes, PR, and captain-merge path. - For task-scoped notes, inspect the item with `bin/fm-tasks-axi.sh show --full`, classify the change as new, duplicate, superseding, or obsolete, then use a considered replacement body through `bin/fm-tasks-axi.sh update --body-file `. Use `--archive-body` when recoverability matters. diff --git a/AGENTS.md b/AGENTS.md index acf9506529c..1757b3b0623 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -284,11 +284,12 @@ Route durable knowledge to its most specific owner: - Captain preferences shared across secondmate domains belong in the primary home's `data/captain-shared.md` under the `secondmate-provisioning` contract. - Fleet-local operational facts belong in curated, home-local `data/learnings.md`. - Task-scoped notes belong with the backlog item, and investigation findings belong in the scout report. -- Knowledge useful to almost every contributor to one project belongs in that project's committed `AGENTS.md`. +- Knowledge useful to almost every contributor to one project belongs in that project's committed `AGENTS.md`, which only deliberate human edits extend. - Knowledge general to every firstmate user belongs in this repo's shared tracked surface. Firstmate never writes a project's `AGENTS.md` directly. -A crewmate creates or updates it lazily through the project's selected delivery path, using `bin/fm-ensure-agents-md.sh` and preferring pointers to authoritative sources over copied detail. +A crewmate edits a project's `AGENTS.md` or `CLAUDE.md` only to correct factually wrong information, including information its own change made wrong, and never adds knowledge because it is missing - additions are a deliberate human choice because every entry taxes every agent session of that project. +A correction edits only the wrong text and never runs `bin/fm-ensure-agents-md.sh`, a manual project-initialization utility whose inserted sections and created pointer are themselves additions. Keep fleet delivery posture and captain-private strategy out of project memory. When the captain invokes `/stow`, load the `stow` skill for its memory curation, knowledge routing, and persistence of the open work records this session is holding; it files and corrects only the open work that session is holding, and never reconciles the backlog against repository or PR reality. diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index 3b6797eb224..cd358243c26 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -98,11 +98,12 @@ # Every scaffold also carries the steering-inbox receive-and-ack section: # process state/.inbox/*.msg in order and acknowledge each by moving it to # handled/ (record, doorbell, and ladder owned by bin/fm-task-inbox-lib.sh). -# Ship tasks include a project-memory section so durable project-intrinsic -# learnings can be committed to AGENTS.md through the project's delivery path; -# it carries the AGENTS.md authoring bar (widely useful knowledge only, pointers -# over copied detail) and defers self-governance recognition and insertion to -# fm-ensure-agents-md.sh's contract. +# Ship tasks include a project-memory section bounding crewmate edits to a +# project's AGENTS.md/CLAUDE.md: only corrections of factually wrong +# information, including wrong information the task itself introduced - never +# additions of missing knowledge. A correction edits only the wrong text and +# never runs fm-ensure-agents-md.sh, whose inserted sections and created +# pointer file are themselves additions. # Scaffolds carry no role scope: fm-spawn.sh supplies fm_brief_worker_role from # fm-dod-lib.sh to every ship/scout launch brief, so this file never becomes a # second owner of a contract that must stay current across relaunches. @@ -650,11 +651,8 @@ $SHARED_INFRA_RULE $INBOX_SECTION # Project memory -If \`AGENTS.md\` or \`CLAUDE.md\` already exists, or if this task produced durable project-intrinsic knowledge, run \`$FM_ROOT/bin/fm-ensure-agents-md.sh .\` in the worktree. -Record only project knowledge useful to almost every future session. -For anything the codebase already shows, prefer a pointer to the authoritative file, command, or doc over copying the detail. -If you touch a project \`AGENTS.md\`, follow \`$FM_ROOT/bin/fm-ensure-agents-md.sh\`'s self-governance contract in the same pass. -Keep it proportionate: skip \`AGENTS.md\` edits for trivial tasks that produced no durable project knowledge. +A project's \`AGENTS.md\` or \`CLAUDE.md\` is loaded into every agent session in that project, so edit it only to correct information that is factually wrong - including information your own change made wrong - and never to add knowledge because it is missing. +A correction edits only the wrong text: do not run \`$FM_ROOT/bin/fm-ensure-agents-md.sh\`, create either file, or add sections, headings, or pointers alongside it. $DOD EOF diff --git a/bin/fm-ensure-agents-md.sh b/bin/fm-ensure-agents-md.sh index b164b5d2137..7b5b4d51d68 100755 --- a/bin/fm-ensure-agents-md.sh +++ b/bin/fm-ensure-agents-md.sh @@ -23,8 +23,10 @@ # filesystem (issue #389). The real-file pointer also eliminates the old # uppercase-literal-target dangling-symlink hazard that a CLAUDE.md -> AGENTS.md # link would have carried for that same mismatch. -# This is a worktree utility for crewmates, not a supervision script, so it does -# not call fm-guard.sh. +# This is a manual project-initialization utility, not a supervision script, +# so it does not call fm-guard.sh. No brief calls it: the sections it inserts +# and the pointer it creates are additions, and AGENTS.md section 6 bounds +# crewmate edits of project memory files to correcting the wrong text only. # Usage: fm-ensure-agents-md.sh [repo-or-worktree-dir] set -eu @@ -109,7 +111,7 @@ write_skeleton() { This file is the project's committed home for project-intrinsic agent knowledge: build, test, release, architecture, and sharp-edge notes that should travel with the code. -- Add durable project-specific notes here as they are discovered through real work. +- Correct entries that work proves wrong; add new ones only by deliberate maintainer choice, never as routine task output. EOF ensure_maintenance_section } diff --git a/docs/architecture.md b/docs/architecture.md index 7cd1aa7a218..1b39f42b138 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -456,16 +456,13 @@ The [Relay configuration reference](configuration.md#promised-public-replies-sta ## Project memory belongs to projects -Durable project-intrinsic agent knowledge lives in each project's committed `AGENTS.md`, with `CLAUDE.md` as a real `@AGENTS.md` import pointer. -Ship briefs prompt crewmates to create or update those files through the normal delivery path; `data/projects.md` stays a thin private registry. -Each project `AGENTS.md` carries self-governance guidance; [`bin/fm-ensure-agents-md.sh`](../bin/fm-ensure-agents-md.sh) owns the canonical wording and idempotent insertion, while its header and help document the explicit mark for equivalent project-owned guidance. -It refuses a case-variant real memory file such as a lowercase `agents.md`, so the pointer's `@AGENTS.md` import resolves to a real `AGENTS.md` on a case-sensitive filesystem, and surfaces the mismatch for manual reconciliation. -The full ownership rule - what is project-intrinsic versus fleet-private, and how firstmate keeps the two apart without writing into project clones - is owned by [`AGENTS.md`](../AGENTS.md) (project and knowledge management). +Project-memory ownership and the crewmate corrections-only boundary are defined in [`AGENTS.md` section 6](../AGENTS.md#6-project-and-knowledge-management); `data/projects.md` stays a thin private registry. +For manual project initialization, [`bin/fm-ensure-agents-md.sh`](../bin/fm-ensure-agents-md.sh) owns the `CLAUDE.md` pointer, self-governance insertion, and case-variant file refusal; its header and help document the explicit mark for equivalent project-owned guidance. ## Operational memory routing `/stow` sweeps the current session for durable knowledge that only exists in conversation and routes each finding to the most specific disk home. -Home-domain captain preferences go to `data/captain.md`, cross-domain shared captain preferences go to the primary home's `data/captain-shared.md`, fleet-local operational facts and gotchas go to home-local `data/learnings.md`, project-intrinsic knowledge goes through normal crewmate delivery into that project's committed `AGENTS.md`, and task-scoped notes or undone next steps go to the backlog. +The destination for each kind of knowledge, including project-intrinsic knowledge, is owned by [`AGENTS.md` section 6](../AGENTS.md#6-project-and-knowledge-management). Memory writes use inspect-then-update rather than blind append; the internal [`stow` skill](../.agents/skills/stow/SKILL.md) owns tier markers, decay, cold archival, and offload. The same pass also persists open-work record state the session is holding - filing a thread that was never recorded and correcting one the session knows went stale - bounded to the open work that session is actually holding. It is deliberately not a reconciliation of durable records against repository or PR reality: its input is the volatile context, so it can only preserve what the session still knows, and no reconciliation that outlives a session exists today. diff --git a/docs/scripts.md b/docs/scripts.md index 68e1072082d..b29076e6d74 100644 --- a/docs/scripts.md +++ b/docs/scripts.md @@ -43,7 +43,7 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | `fm-herdr-ci-cleanup.sh` | Snapshot and tear down only job-owned `fm-lab-*` sessions in the Herdr CI lane | | `fm-test-run.sh` | Behavior-test runner: selection, portable lanes, bounded concurrency, budgets, coverage guard, timing/JSON; refuses to execute in the repository primary checkout when `FM_TASK_ID` marks a task worker | | `fm-test-isolation-proof.sh` | Concurrent isolation harness and portable candidate set owner | -| `fm-ensure-agents-md.sh` | Ensure a project's real `AGENTS.md`, its `CLAUDE.md` `@AGENTS.md` pointer, and self-governance guidance (explicit project mark documented in the helper's header and help) | +| `fm-ensure-agents-md.sh` | Manually initialize project agent-memory files (see the helper's header and help) | | `fm-guard.sh` | Warn on primary-checkout tangles, main-session pending wakes, and unhealthy supervision | | `fm-primary-scope-lib.sh` | Shared marker-or-plain-checkout primary-home predicate for tracked hooks | | `fm-session-lock-lib.sh` | Shared session-lock ownership from harness ancestry or a trusted Claude session id for fm-lock.sh and the Claude Stop auto-arm, plus the read-only lock inspection behind `fm-lock.sh status` and `fm-inbox.sh ready` | diff --git a/tests/fm-brief.test.sh b/tests/fm-brief.test.sh index 418dd3a33ca..cc8cdc23c2d 100755 --- a/tests/fm-brief.test.sh +++ b/tests/fm-brief.test.sh @@ -484,6 +484,10 @@ test_ask_user_escalation_format() { pass "fm-brief.sh: no-mistakes ask-user findings use one event plus a verbatim snapshot" } +# The project-memory section bounds crewmate edits of a project's AGENTS.md or +# CLAUDE.md to corrections of factually wrong information - including wrong +# information the task itself introduced - and never invites additions of +# missing knowledge, because those files tax every agent session of the project. test_ship_project_memory_wording() { local home id brief home="$TMP_ROOT/project-memory-home" @@ -492,13 +496,19 @@ test_ship_project_memory_wording() { FM_HOME="$home" "$ROOT/bin/fm-brief.sh" "$id" some-proj --mode no-mistakes >/dev/null 2>&1 brief="$home/data/$id/brief.md" assert_present "$brief" "brief was not scaffolded" - assert_grep "Record only project knowledge useful to almost every future session." "$brief" \ - "project-memory contract lost the durable-knowledge bar" - assert_grep "prefer a pointer to the authoritative file, command, or doc over copying the detail" "$brief" \ - "project-memory contract lost pointer-over-copy guidance" - assert_grep "follow \`$ROOT/bin/fm-ensure-agents-md.sh\`'s self-governance contract" "$brief" \ - "project-memory contract no longer defers to the ensure helper" - pass "fm-brief.sh: ship project-memory wording carries the AGENTS.md authoring bar" + assert_grep "loaded into every agent session" "$brief" \ + "project-memory contract lost the per-session cost rationale" + assert_grep "only to correct information that is factually wrong" "$brief" \ + "project-memory contract lost the corrections-only bound" + assert_grep "including information your own change made wrong" "$brief" \ + "project-memory contract lost the self-inflicted correction case" + assert_grep "never to add knowledge because it is missing" "$brief" \ + "project-memory contract still permits additions of missing knowledge" + assert_no_grep "if this task produced durable project-intrinsic knowledge" "$brief" \ + "project-memory contract still invites additions for durable knowledge" + assert_grep "A correction edits only the wrong text: do not run \`$ROOT/bin/fm-ensure-agents-md.sh\`" "$brief" \ + "project-memory contract no longer forbids the ensure helper on a correction" + pass "fm-brief.sh: ship project-memory wording bounds edits to corrections of wrong information" } test_herdr_lab_contract_is_explicit_and_complete() { From ca8c293e65176cf07c4caa2bb7517bf044a1ce10 Mon Sep 17 00:00:00 2001 From: Kun Chen <3233006+kunchenguid@users.noreply.github.com> Date: Thu, 24 Sep 2026 23:05:32 -0700 Subject: [PATCH 07/84] feat: permit gate lifecycle calls against disposable lab homes (#5635) * fix(bin): let gate agents drive lifecycle against marked lab homes Part 2 of the #5615 split. A no-mistakes gate agent runs inside a checkout carrying the fleet-captain identity, so fm-gate-refuse-lib refuses fleet mutation on the gate signal. That refusal was absolute, which kept gate validation from ever exercising the real lifecycle. Stamp a disposable lab FM_HOME with a .fm-lab-home marker file that only bin/fm-lab-home.sh writes, and only onto a fresh empty dir, so no call path can mark a populated real home. fm_refuse_if_gate_agent then permits lifecycle only when FM_HOME carries the marker and is driven through its stock layout - any FM_*_OVERRIDE relocation stays refused so part of the "lab" cannot be split back onto the real fleet. The threat model is a confused agent touching the real fleet, not deliberate forgery, so the marker is a plain token file rather than a bound record. FM_GATE_REFUSE_BYPASS is unchanged: it still serves the test harness, which cannot mark hundreds of temp homes. Teardown's slot-ownership scan compared state-dir paths textually while fm_firstmate_root_home canonicalizes, so a lab home under a symlinked TMPDIR scanned its own record twice and self-collided; compare file identity (-ef) instead. * no-mistakes(review): Refuse unlistable lab homes and hardlinked slot records * no-mistakes(review): Mint lab markers only on verified-empty fresh dirs * no-mistakes(document): Clarify lab-home gate documentation and comment contracts * no-mistakes(document): Clarify lab-home gate documentation and remove stale claims * no-mistakes(document): Clarify gate lab-home documentation and boundary wording --- .no-mistakes.yaml | 3 +- bin/fm-gate-refuse-lib.sh | 95 +++++++++++++---- bin/fm-lab-home.sh | 47 +++++++++ bin/fm-teardown.sh | 6 +- docs/architecture.md | 6 +- docs/scripts.md | 5 +- tests/fm-gate-refuse.test.sh | 120 ++++++++++++++++++++++ tests/fm-teardown-endpoint-safety.test.sh | 17 +++ 8 files changed, 272 insertions(+), 27 deletions(-) create mode 100755 bin/fm-lab-home.sh diff --git a/.no-mistakes.yaml b/.no-mistakes.yaml index 10a1c9bef28..53f7ecb72a8 100644 --- a/.no-mistakes.yaml +++ b/.no-mistakes.yaml @@ -5,7 +5,7 @@ # no-mistakes review/fix/document/test/lint/pr/rebase/ci agent never adopts that # identity or drives the fleet. Trusted-only: a pushed branch cannot turn this off, # so it is honored only from the default-branch copy of this file. Layered above -# the NO_MISTAKES_GATE lifecycle refusal (bin/fm-gate-refuse-lib.sh) and the +# gate-context lifecycle boundary (bin/fm-gate-refuse-lib.sh) and the # HEAD-continuity guard; see docs/architecture.md "No-mistakes gate authority boundary." disable_project_settings: true @@ -40,6 +40,7 @@ test: Run live Herdr scenarios only through bin/fm-herdr-lab.sh with a named non-default fm-lab-* session, following that helper's prepare, provision, run, and teardown contract exactly. Never touch the live default Herdr session or fleet panes. Prefer a throwaway lab for spawn, long-launch, and Claude-path proofs, and tear it down in the same evidence turn. + Lifecycle calls against a throwaway firstmate home proceed inside the gate only when the home was minted by `bin/fm-lab-home.sh create ` and driven as plain `FM_HOME=` with no FM_*_OVERRIDE relocations; every other home stays refused. Do not mutate the operator primary checkout, real fleet FM_HOME state, or production credentials, and keep git changes otherwise inside the run worktree. Read docs/herdr-backend.md and the bin/fm-herdr-lab.sh header as the owners of Herdr lab mechanics rather than reproducing that manual here. Ship or scout briefs that will drive Herdr lifecycle still require --herdr-lab at scaffold time; these Test-agent instructions are not a substitute for that brief flag. diff --git a/bin/fm-gate-refuse-lib.sh b/bin/fm-gate-refuse-lib.sh index 8e624408a5e..de22674411e 100644 --- a/bin/fm-gate-refuse-lib.sh +++ b/bin/fm-gate-refuse-lib.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash -# fm-gate-refuse-lib.sh - fail-closed refusal that keeps a no-mistakes GATE agent -# out of firstmate's fleet lifecycle. +# fm-gate-refuse-lib.sh - refuse no-mistakes gate lifecycle calls against the +# real fleet while allowing marked disposable lab homes. # # The hazard (data/nm-gate-ambient-authority-containment-c3/report.md): a # no-mistakes gate agent runs inside a firstmate checkout with a free shell, so @@ -11,12 +11,12 @@ # # no-mistakes owns the authority-removal half (it neutralizes the project # instructions and stamps NO_MISTAKES_GATE into the gate agent's environment). -# THIS is the firstmate capability-removal half: an enforceable script refusal, -# not a prose rule the neutralized agent would never read. It is sourced at the -# top of the three fleet-lifecycle entrypoints and called before any fleet -# mutation, so a gate agent that still reaches for the fleet is stopped cold. +# THIS is the firstmate capability boundary: an enforceable script check, +# not a prose rule the neutralized agent would never read. It is sourced by the +# four fleet-lifecycle entrypoints and called before their fleet mutation, so +# a gate agent that reaches for the real fleet is stopped cold. # -# Two independent signals, either of which refuses (fail closed): +# Two independent gate-context signals, either of which triggers the check: # # 1. NO_MISTAKES_GATE set - the durable env marker no-mistakes stamps into every # gate agent. This is the primary signal and covers a relocated NM_HOME. @@ -24,7 +24,7 @@ # repo (.../.no-mistakes/repos/*.git) - the UNSPOOFABLE backstop. It derives # from the checkout's real filesystem location, which the agent cannot # relocate without breaking the gate's own git operations, so it still -# refuses even if the agent tampered NO_MISTAKES_GATE away. Its limit: the +# detects a gate even if the agent tampered NO_MISTAKES_GATE away. Its limit: the # literal-path match only fires for the default NM_HOME (~/.no-mistakes); a # relocated NM_HOME is covered by signal 1. # @@ -32,13 +32,27 @@ # crew worktree - has NEITHER signal and is COMPLETELY unaffected: the function # returns 0 and the lifecycle proceeds exactly as before. # -# This mirrors the unspoofable-marker precedent in bin/fm-marker-lib.sh: a signal -# the agent cannot forge, keyed on at a chokepoint, keeping the pattern familiar -# to firstmate maintainers. It layers ABOVE no-mistakes' separately-shipping -# HEAD-continuity guard, which remains the adversarial/residual backstop. +# THE ONE AUTHORIZED EXCEPTION - a disposable lab home: a gate agent may drive +# lifecycle against an FM_HOME that carries the FM_GATE_LAB_MARKER file, because +# bin/fm-lab-home.sh stamps it only on an empty directory +# (fm_gate_lab_mark refuses a populated dir, so the helper cannot mark a real home). +# The allowance additionally requires every FM_*_OVERRIDE to be empty or unset, +# so the lab call uses the marked home's stock layout and no override can split +# part of the "lab" back onto the real fleet. The threat model stays a CONFUSED +# agent: a hostile agent that would hand-forge the marker file is the +# adversarial case no-mistakes' neutral-execution-context and the +# HEAD-continuity guard already own, so the check is a plain token file, not a +# bound record. This is an allowance on the CAPABILITY side only: +# fm_is_gate_agent still reports the gate context, so the sessionstart +# stand-downs that read it directly are unaffected by the marker. +# +# The gate-context backstop mirrors the unspoofable-marker precedent in +# bin/fm-marker-lib.sh; the lab-home marker is deliberately not unspoofable. +# This boundary layers above no-mistakes' separately-shipping HEAD-continuity +# guard, which remains the adversarial/residual backstop. # # TEST-HARNESS ESCAPE HATCH (FM_GATE_REFUSE_BYPASS=1): firstmate's own test suite -# must exercise the REAL fm-spawn/fm-send/fm-teardown, but the no-mistakes gate +# must exercise the real fleet entrypoints, but the no-mistakes gate # runs that suite FROM a gate worktree (cwd git-common-dir under # .no-mistakes/repos/*.git, and possibly NO_MISTAKES_GATE set) - the exact # environment this guard refuses. So both signals would fire during firstmate's @@ -53,16 +67,52 @@ # neutral-execution-context and the HEAD-continuity guard. The dedicated # tests/fm-gate-refuse.test.sh strips the bypass so it still verifies real refusal. # -# Sourced by bin/fm-spawn.sh, bin/fm-send.sh, bin/fm-teardown.sh, -# bin/fm-sessionstart-nudge.sh, and the tests. +# Sourced by the fleet lifecycle entrypoints, session-start hooks, +# bin/fm-lab-home.sh, and the tests. # No side effects on source. set -u / set -e safe. The refusal is a hard exit, -# not a return, because there is no safe way to continue a fleet mutation from a -# gate context. +# not a return, because an unpermitted gate call cannot safely mutate the fleet. # The exit code every refusal uses, distinct enough to recognize in a caller or # test as "the gate refusal fired" rather than an ordinary usage error. FM_GATE_REFUSE_EXIT=3 +# The disposable-lab-home marker file and the token line it must carry. The +# format is owned here; bin/fm-lab-home.sh is the supported writer. +FM_GATE_LAB_MARKER='.fm-lab-home' +FM_GATE_LAB_TOKEN='fm-lab-home v1' + +# fm_gate_lab_home : return 0 when is a marked disposable lab home. +fm_gate_lab_home() { + local home=${1:-} + [ -n "$home" ] || return 1 + [ -f "$home/$FM_GATE_LAB_MARKER" ] || return 1 + [ "$(sed -n '1p' "$home/$FM_GATE_LAB_MARKER" 2>/dev/null || true)" = "$FM_GATE_LAB_TOKEN" ] +} + +# fm_gate_lab_mark : stamp as a disposable lab home. Fails closed on +# any dir that is not empty, so this can never mark a populated real home. +fm_gate_lab_mark() { + local home=${1:-} listing + [ -n "$home" ] && [ -d "$home" ] || return 1 + listing=$(find "$home" -mindepth 1 -maxdepth 1 -print -quit 2>/dev/null) || return 1 + [ -z "$listing" ] || return 1 + printf '%s\n' "$FM_GATE_LAB_TOKEN" > "$home/$FM_GATE_LAB_MARKER" +} + +# fm_gate_lab_permitted: return 0 when the current call targets a marked lab +# home through a stock layout - $FM_HOME carries the marker and no +# FM_*_OVERRIDE relocation has a nonempty value. +fm_gate_lab_permitted() { + local v + fm_gate_lab_home "${FM_HOME:-}" || return 1 + for v in "${!FM_@}"; do + case "$v" in + *_OVERRIDE) [ -z "${!v}" ] || return 1 ;; + esac + done + return 0 +} + # fm_is_gate_agent: return 0 without output when this process looks like a # no-mistakes gate agent. An optional root anchors the git-common-dir check; # callers that omit it retain the historical current-worktree behavior. @@ -88,11 +138,16 @@ fm_is_gate_agent() { } # fm_refuse_if_gate_agent: exit FM_GATE_REFUSE_EXIT with a clear stderr message if -# this process looks like a no-mistakes gate agent. Call before any fleet -# mutation. No-ops (returns 0) for a normal firstmate session, or when firstmate's -# own test harness sets FM_GATE_REFUSE_BYPASS=1 (see the header). +# this process looks like a no-mistakes gate agent without a permitted lab home. +# Call before any fleet mutation. No-ops (returns 0) for a normal firstmate +# session, a permitted lab home, or when firstmate's own test harness sets +# FM_GATE_REFUSE_BYPASS=1 (see the header). fm_refuse_if_gate_agent() { fm_is_gate_agent "${1:-.}" || return 0 + if fm_gate_lab_permitted; then + echo "fm-gate-refuse: gate agent lifecycle permitted only against lab home $FM_HOME" >&2 + return 0 + fi if [ "$FM_GATE_REFUSE_REASON" = env ]; then echo "error: no-mistakes gate agent must not drive the fleet (NO_MISTAKES_GATE set)" >&2 else diff --git a/bin/fm-lab-home.sh b/bin/fm-lab-home.sh new file mode 100755 index 00000000000..113a0c8797e --- /dev/null +++ b/bin/fm-lab-home.sh @@ -0,0 +1,47 @@ +#!/usr/bin/env bash +# fm-lab-home.sh - mint a disposable firstmate "lab" home. +# +# A lab home is a throwaway FM_HOME that a no-mistakes GATE agent may drive +# through the fleet lifecycle entrypoints: bin/fm-gate-refuse-lib.sh refuses +# those calls inside a gate agent unless FM_HOME carries the marker file this +# helper writes (the lib owns the marker format and authorization decision; +# this script is the supported writer). +# +# Usage: +# fm-lab-home.sh create make a marked lab home and print it; +# refused on any existing non-empty dir +# +# A lab home is the stock layout only - state/, data/, config/, projects/ - and +# callers remove it with ordinary rm -rf when done. Drive it with plain +# FM_HOME=; any FM_*_OVERRIDE relocation defeats the allowance. +set -u + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +# shellcheck source=bin/fm-gate-refuse-lib.sh +. "$SCRIPT_DIR/fm-gate-refuse-lib.sh" + +fm_lab_home_error() { + echo "fm-lab-home: $*" >&2 +} + +case "${1:-}" in + create) + dir=${2:-} + [ -n "$dir" ] || { fm_lab_home_error "create requires a directory path"; exit 2; } + if [ -e "$dir" ] && [ ! -d "$dir" ]; then + fm_lab_home_error "refusing '$dir': exists and is not a directory" + exit 1 + fi + mkdir -p "$dir" || exit 1 + fm_gate_lab_mark "$dir" || { + fm_lab_home_error "refusing '$dir': a lab marker is only ever stamped on a fresh empty dir" + exit 1 + } + mkdir -p "$dir/state" "$dir/data" "$dir/config" "$dir/projects" || exit 1 + printf '%s\n' "$dir" + ;; + *) + fm_lab_home_error "usage: fm-lab-home.sh create " + exit 2 + ;; +esac diff --git a/bin/fm-teardown.sh b/bin/fm-teardown.sh index a1cf67c220a..26f5bb84709 100755 --- a/bin/fm-teardown.sh +++ b/bin/fm-teardown.sh @@ -2295,7 +2295,11 @@ require_exclusive_worktree_slot_record() { for state_dir in "${TREEHOUSE_OWNER_STATES[@]}"; do for other in "$state_dir"/*.meta; do [ -f "$other" ] && [ ! -L "$other" ] || continue - [ "$other" != "$record_meta" ] || continue + # Identity, not spelling: the same record reached through a differently + # resolved state dir (e.g. a symlinked $FM_HOME) is still this record. A + # differently named hardlink is another task's record, so the name must + # match too. + [ "${other##*/}" = "${record_meta##*/}" ] && [ "$other" -ef "$record_meta" ] && continue other_id=$(basename "$other" .meta) for field in worktree home; do other_path=$(fm_meta_get "$other" "$field") diff --git a/docs/architecture.md b/docs/architecture.md index 1b39f42b138..f5547c189f9 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -291,9 +291,9 @@ Placement is proven only at launch, so `bin/fm-spawn.sh` also exports the task i Firstmate's own no-mistakes gate runs agents inside a checkout that also contains the fleet-captain identity in `AGENTS.md`, so gate execution needs an authority boundary separate from ordinary crewmate worktree isolation. The tracked `.no-mistakes.yaml` sets `disable_project_settings: true`; no-mistakes honors that setting only from the trusted default-branch copy, so a pushed branch cannot enable its own project instructions during validation. -Independently, `fm-spawn.sh`, `fm-send.sh`, `fm-control.sh`, and `fm-teardown.sh` source `bin/fm-gate-refuse-lib.sh` and exit with status 3 before fleet mutation when the gate environment marker is present or the current checkout matches the default no-mistakes gate-repository topology. -A normal primary checkout or crewmate worktree has neither signal and remains unaffected. -The helper's header owns the exact signal detection, relocated-home limitation, test-harness bypass, and relationship to no-mistakes' HEAD-continuity guard. +Independently, the fleet lifecycle entrypoints use `bin/fm-gate-refuse-lib.sh` to refuse gate calls against the real fleet, while permitting validation against a disposable lab home minted by `bin/fm-lab-home.sh`. +A normal primary checkout or crewmate worktree remains unaffected. +The refusal library's header owns the gate detection, lab-home exception, test-harness bypass, and relationship to no-mistakes' HEAD-continuity guard; the lab helper's header owns its usage. ## Two task shapes diff --git a/docs/scripts.md b/docs/scripts.md index b29076e6d74..6bb9b68e1b5 100644 --- a/docs/scripts.md +++ b/docs/scripts.md @@ -3,7 +3,7 @@ The first mate drives these; interactive entrypoints work by hand too, while `*-lib.sh` files are sourced helpers. Each row is one purpose clause only: the script's own header comment is the authoritative description of its behavior, flags, and contracts, so read the header before first use. If you have changed away from the firstmate home in an interactive shell, invoke these scripts by absolute path through the repo's `bin/` directory; the scripts self-locate internally after they start. -The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarized in [architecture.md](architecture.md#no-mistakes-gate-authority-boundary), while `docs/sessionstart-nudge.md` covers the silent session-open hook use; `fm-gate-refuse-lib.sh`'s header owns its exact contract. +The shared no-mistakes gate lifecycle boundary is summarized in [architecture.md](architecture.md#no-mistakes-gate-authority-boundary), while `docs/sessionstart-nudge.md` covers the silent session-open hook use; `fm-gate-refuse-lib.sh`'s header owns its exact contract. | Script | Purpose | | ------------------------ | ------------------------------------------------------------------------------------ | @@ -38,6 +38,7 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | `fm-brief-heading-lib.sh` | Single owner of reading a brief's sections, shared by the `--intent` contract, spawn and promotion validation, and `fm-dispatch-resolve.sh` | | `fm-herdr-lab.sh` | Provision and guardedly operate an isolated, never-default Herdr lab session | | `fm-herdr-lab-viewer.py` | The pty engine behind `fm-herdr-lab.sh viewer`: one real foreground Herdr client on a non-zero window grid | +| `fm-lab-home.sh` | Mint a disposable lab home for gate lifecycle validation | | `fm-install-herdr.sh` | Install CI's exact-version Herdr pin with official asset URL, SHA-256, and protocol checks | | `fm-install-treehouse.sh`| Install CI's exact-version Treehouse pin for real-Herdr E2E that needs spawn worktrees | | `fm-herdr-ci-cleanup.sh` | Snapshot and tear down only job-owned `fm-lab-*` sessions in the Herdr CI lane | @@ -85,7 +86,7 @@ The shared no-mistakes gate refusal for fleet lifecycle entrypoints is summarize | `fm-procevent-remote-reply.sh` | Relay the remote-secondmate status stream through non-destructive process-event deltas | | `fm-procevent-quota.sh` | Wake Firstmate when tracked quota drops below a threshold, is exhausted, or cannot be polled | | `fm-procevent-when.sh` | Fire a trust-bound deterministic action at most once when its registered condition holds, then wake with the outcome | -| `fm-gate-refuse-lib.sh` | Shared no-mistakes gate-context refusal for fleet lifecycle entrypoints | +| `fm-gate-refuse-lib.sh` | Shared gate-context lifecycle boundary for real and lab homes | | `fm-watch-arm.sh` | Verified home-scoped watcher arm wrapper with loud cycle endings and bounded lifecycle ledger | | `fm-watch-checkpoint.sh` | Run one bounded foreground watcher checkpoint for Codex-style supervision | | `fm-watch.sh` | Singleton-safe watcher: absorb benign wakes, detect stalled local-secondmate wake queues, and exit on actionable ones | diff --git a/tests/fm-gate-refuse.test.sh b/tests/fm-gate-refuse.test.sh index 8ef32c14949..d55fb32c7f1 100755 --- a/tests/fm-gate-refuse.test.sh +++ b/tests/fm-gate-refuse.test.sh @@ -13,6 +13,13 @@ # A normal firstmate session (real primary, real crew worktree) has NEITHER # signal and is completely unaffected. # +# The one authorized exception is a disposable LAB home: bin/fm-lab-home.sh +# stamps a marker file only on a fresh empty dir, and inside a gate context the +# refusal lets a lifecycle call proceed only when FM_HOME is a marked lab home +# used through its stock layout (any FM_*_OVERRIDE relocation stays refused). +# The helper legs below cover the admit/refuse contract; teardown additionally +# proves the admit end-to-end through a real entrypoint. +# # Each entrypoint is exercised in three scenarios, isolating exactly ONE signal: # - env-marker refuse : neutral cwd + NO_MISTAKES_GATE set -> exit 3, no mutation # - path-backstop refuse: gate-worktree cwd + marker UNSET -> exit 3, no mutation @@ -31,6 +38,7 @@ set -u . "$(dirname "${BASH_SOURCE[0]}")/fixtures.sh" GATE_LIB="$ROOT/bin/fm-gate-refuse-lib.sh" +LABHOME="$ROOT/bin/fm-lab-home.sh" SPAWN="$ROOT/bin/fm-spawn.sh" SEND="$ROOT/bin/fm-send.sh" TEARDOWN="$ROOT/bin/fm-teardown.sh" @@ -131,6 +139,82 @@ test_helper_normal_is_noop() { pass "fm-gate-refuse-lib: no-op for a normal session (neither signal, set -eu clean)" } +# --- disposable lab homes ---------------------------------------------------- + +# run_guard_lib_home [ASSIGN...] -> combined output : like +# run_guard_lib but with FM_HOME=; extra ASSIGN args carry the gate +# signal (NO_MISTAKES_GATE=1) or an override to exercise the stock-layout +# requirement. All FM_*_OVERRIDE vars are unset first so the suite stays +# hermetic inside a real gate. +run_guard_lib_home() { + local cwd=$1 home=$2; shift 2 + # shellcheck disable=SC2016 # $1/$2 expand in the child shell, not here. + env -u NO_MISTAKES_GATE -u FM_GATE_REFUSE_BYPASS \ + -u FM_ROOT_OVERRIDE -u FM_STATE_OVERRIDE -u FM_DATA_OVERRIDE \ + -u FM_PROJECTS_OVERRIDE -u FM_CONFIG_OVERRIDE \ + FM_HOME="$home" "$@" \ + bash -c 'cd "$1" || exit 111; set -eu; . "$2"; fm_refuse_if_gate_agent' \ + _ "$cwd" "$GATE_LIB" 2>&1 +} + +test_helper_lab_home_admits() { + local lab plain out rc + lab=$("$LABHOME" create "$TMP/lab-home") || fail "lab-home create failed" + plain="$TMP/plain-home"; mkdir -p "$plain" + + # gate + marked lab home -> permitted (env signal; the path backstop shares + # the same fm_is_gate_agent gate). + out=$(run_guard_lib_home "$NORMAL_CWD" "$lab" NO_MISTAKES_GATE=1); rc=$? + expect_code 0 "$rc" "helper: gate + marked lab home must be permitted" + assert_contains "$out" "lab home" "helper: lab permit should name the lab home" + + # gate + unmarked home -> refused. + out=$(run_guard_lib_home "$NORMAL_CWD" "$plain" NO_MISTAKES_GATE=1); rc=$? + expect_code 3 "$rc" "helper: gate + unmarked home must still refuse" + assert_contains "$out" "$ENV_MSG" "helper: unmarked-home refusal message" + + # gate + marked lab + an FM_*_OVERRIDE -> refused: the allowance requires + # the stock layout so an override cannot split state onto the real fleet. + out=$(run_guard_lib_home "$NORMAL_CWD" "$lab" NO_MISTAKES_GATE=1 FM_STATE_OVERRIDE="$lab/state"); rc=$? + expect_code 3 "$rc" "helper: lab home driven through FM_STATE_OVERRIDE must refuse" + assert_contains "$out" "$ENV_MSG" "helper: override refusal message" + + # no gate signal + lab home -> still a normal no-op. + out=$(run_guard_lib_home "$NORMAL_CWD" "$lab"); rc=$? + expect_code 0 "$rc" "helper: lab home outside a gate must not refuse" + pass "fm-gate-refuse-lib: marked lab home permitted in a gate; unmarked home or an override stay refused" +} + +test_lab_home_helper() { + local lab populated unlistable newline out rc + # create on an absent path mints the marker and the stock layout. + lab=$("$LABHOME" create "$TMP/lab-new"); rc=$? + expect_code 0 "$rc" "lab-home: create must succeed on a fresh path" + assert_present "$lab/.fm-lab-home" "lab-home: create must write the marker" + for d in state data config projects; do + [ -d "$lab/$d" ] || fail "lab-home: missing stock dir $d" + done + out=$("$LABHOME" create "$lab" 2>&1); rc=$? + [ "$rc" -ne 0 ] || fail "lab-home: create on an existing lab home must refuse" + # refuses a populated dir and leaves it unmarked. + populated="$TMP/populated"; mkdir -p "$populated/state"; echo x > "$populated/state/x.meta" + out=$("$LABHOME" create "$populated" 2>&1); rc=$? + [ "$rc" -ne 0 ] || fail "lab-home: create on a populated dir must refuse" + assert_absent "$populated/.fm-lab-home" "lab-home: refused create must not write the marker" + # refuses a populated dir it cannot list, rather than reading it as empty. + unlistable="$TMP/unlistable"; mkdir -p "$unlistable/state"; chmod 300 "$unlistable" + out=$("$LABHOME" create "$unlistable" 2>&1); rc=$? + chmod 700 "$unlistable" + [ "$rc" -ne 0 ] || fail "lab-home: create on an unlistable dir must refuse" + assert_absent "$unlistable/.fm-lab-home" "lab-home: unlistable create must not write the marker" + # refuses a dir whose only entry has a newline-only name. + newline="$TMP/newline-entry"; mkdir -p "$newline/"$'\n' + out=$("$LABHOME" create "$newline" 2>&1); rc=$? + [ "$rc" -ne 0 ] || fail "lab-home: create on a dir holding a newline-named entry must refuse" + assert_absent "$newline/.fm-lab-home" "lab-home: newline-entry create must not write the marker" + pass "fm-lab-home: create mints marked stock homes only on fresh empty dirs; anything else is refused" +} + # --- fm-spawn --------------------------------------------------------------- # run_spawn [ASSIGN...] -> combined output @@ -321,6 +405,25 @@ run_teardown() { "$TEARDOWN" task-x1 ) 2>&1 } +# run_teardown_lab [ASSIGN...] -> combined output +# Lab-home counterpart of run_teardown: FM_HOME=, no FM_*_OVERRIDE. +run_teardown_lab() { + local cwd=$1 case_dir=$2; shift 2 + ( cd "$cwd" && env -u NO_MISTAKES_GATE -u FM_GATE_REFUSE_BYPASS \ + "FM_HOME=$case_dir" \ + "PATH=$case_dir/fakebin:$PATH" "$@" \ + "$TEARDOWN" task-x1 ) 2>&1 +} + +# make_teardown_lab_case -> echoes a marked lab case dir holding the same +# landed task as make_teardown_case (the marker is stamped while the dir is +# still empty, then the fixture populates it). +make_teardown_lab_case() { + local name=$1 + "$LABHOME" create "$TMP/$name" >/dev/null || return 1 + make_teardown_case "$name" +} + test_teardown_refuses_and_admits() { local case_dir out rc @@ -345,6 +448,21 @@ test_teardown_refuses_and_admits() { assert_not_contains "$out" "$ENV_MSG" "teardown: normal teardown must not print the gate refusal" assert_not_contains "$out" "$PATH_MSG" "teardown: normal teardown must not print the backstop refusal" assert_not_contains "$out" "REFUSED" "teardown: normal teardown of landed work must not refuse" + + # lab-home admit: gate context + FM_HOME=marked lab home -> tears down. + case_dir=$(make_teardown_lab_case teardown-lab) + out=$(run_teardown_lab "$GATE_WT" "$case_dir"); rc=$? + expect_code 0 "$rc" "teardown: gate + marked lab home must tear down landed work" + assert_absent "$case_dir/state/task-x1.meta" "teardown: lab teardown should remove the task record" + + # regression: a home reached through a symlinked spelling must not + # self-collide in the slot-ownership scan - the canonical root home and the + # textual state dir resolve to the same record by identity, not path bytes. + case_dir=$(make_teardown_case teardown-symlink) + ln -s "$case_dir" "$TMP/teardown-symlinked" + out=$(run_teardown_lab "$NORMAL_CWD" "$TMP/teardown-symlinked"); rc=$? + expect_code 0 "$rc" "teardown: a symlinked home spelling must not self-collide" + assert_absent "$case_dir/state/task-x1.meta" "teardown: symlinked-home teardown should remove the task" pass "fm-teardown: refuses on marker and gate-worktree backstop; a normal teardown is unaffected" } @@ -352,6 +470,8 @@ test_helper_env_marker_refuses test_helper_empty_env_marker_refuses test_helper_path_backstop_refuses test_helper_normal_is_noop +test_helper_lab_home_admits +test_lab_home_helper test_spawn_refuses_and_admits test_send_refuses_and_admits test_teardown_refuses_and_admits diff --git a/tests/fm-teardown-endpoint-safety.test.sh b/tests/fm-teardown-endpoint-safety.test.sh index 4002cf4df3e..01dc74239e6 100755 --- a/tests/fm-teardown-endpoint-safety.test.sh +++ b/tests/fm-teardown-endpoint-safety.test.sh @@ -530,6 +530,23 @@ test_reused_pool_slot_refuses_before_touching_the_other_task() { [ ! -s "$dir/runtime.log" ] \ || fail "teardown reached the runtime on a slot held by a secondmate home: $(cat "$dir/runtime.log")" + # A second task record that is a hardlink of this one is still a second + # claim on the slot, not this record reached through another spelling. + dir=$(make_case slot-reuse-hardlink) + mark_case_as_treehouse_pool "$dir" + fm_write_meta "$dir/home/state/$id.meta" \ + "window=firstmate:fm-$id" "endpoint_task_id=$id" \ + "worktree=$dir/worktree" "project=$dir/project" "kind=scout" + ln "$dir/home/state/$id.meta" "$dir/home/state/$other.meta" + set +e + run_case "$dir" "$id" > "$dir/stdout" 2> "$dir/stderr" + rc=$? + set -e + [ "$rc" -ne 0 ] || fail "teardown returned a pool slot a hardlinked second task record still holds" + assert_present "$dir/worktree/sentinel" "teardown reset a pool slot a hardlinked second task record still holds" + assert_contains "$(cat "$dir/stderr")" "$other" \ + "hardlink refusal should name the other task record" + pass "fm-teardown: a pool slot named by a second task record is never returned, killed, or reset" } From b575497a9645ef7e586da08946a682e8f43edb58 Mon Sep 17 00:00:00 2001 From: Christopher McKay <101884182+karotkriss@users.noreply.github.com> Date: Fri, 25 Sep 2026 02:18:55 -0400 Subject: [PATCH 08/84] fix(bin): evict a watcher whose beacon stalls past a hard bound instead of refusing every re-arm (#5594) * fix(bin): replace a watcher whose beacon stalls past a hard bound instead of refusing every re-arm A fleet watcher that is alive but whose liveness beacon has gone stale could never be replaced: every re-arm was refused because the lock holder was a live pid, and the holder was never evicted because it was not dead. Add FM_WATCHER_STALL_BOUND (default 3x the stale grace): below it the refusal is unchanged; at or past it the arm re-verifies the holder against the lock's recorded identity, sends TERM, waits boundedly, and takes the lock the normal way, ledgering a stalled-holder-replaced row. A holder that survives TERM keeps the old refusal. Fixes #4400 * no-mistakes(test): poll for replacement message to fix watcher-lock test flake * no-mistakes(document): document FM_WATCHER_STALL_BOUND in config inventory --- bin/fm-watch-arm.sh | 8 +++++ bin/fm-watch.sh | 48 +++++++++++++++++++++++++++-- docs/configuration.md | 1 + docs/turnend-guard.md | 1 + tests/fm-watcher-lock.test.sh | 58 +++++++++++++++++++++++++++++++++++ 5 files changed, 113 insertions(+), 3 deletions(-) diff --git a/bin/fm-watch-arm.sh b/bin/fm-watch-arm.sh index 31f4a94727e..687252eca7a 100755 --- a/bin/fm-watch-arm.sh +++ b/bin/fm-watch-arm.sh @@ -574,6 +574,14 @@ deadline=$(( $(date +%s) + CONFIRM_TIMEOUT + 1 )) while :; do if healthy_watcher; then if [ "$HEALTHY_PID" = "$child" ]; then + if grep -q '^watcher: replaced stalled pid ' "$child_out" 2>/dev/null; then + # The child evicted a live holder whose beacon stalled past the hard + # bound (bin/fm-watch.sh evict_stalled_holder). Ledger that as its own + # row - lock_before still names the evicted holder - then reopen this + # cycle so its ordinary close row follows as usual. + cycle_log_append 0 none stalled-holder-replaced "started:$child" + cycle_begin "$child" started "$HEALTHY_IDENTITY" + fi cycle_refresh_lock_before if ! handling_generation=$(handling_successor_generation); then cleanup_child diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 4254f6fd4fb..d3de5cea396 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -151,7 +151,13 @@ # FM_SECONDMATE_LIVENESS_WINDOW_SECS) # For normal supervision, resume the session-start primary-harness protocol # after each printed reason. Direct duplicate invocations of this script still -# no-op through the watcher singleton lock. +# no-op through the watcher singleton lock. A live holder whose beacon is stale +# past the grace (FM_WATCHER_STALE_GRACE, default max(300, FM_POLL+60)) is +# refused with "lock held by live pid ... but heartbeat is stale"; one stale past +# the hard bound FM_WATCHER_STALL_BOUND (default 3x that grace) is instead +# evicted with TERM after its recorded identity is re-verified, and this arm +# starts in its place, printing "watcher: replaced stalled pid (...)". A +# holder that survives TERM keeps the refusal and the nonzero exit. set -u SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -258,6 +264,11 @@ POLL=${FM_POLL:-15} # seconds between cycles # This recomputes the library default above now that the real configured # POLL is known. WATCHER_STALE_GRACE=${FM_WATCHER_STALE_GRACE:-${FM_GUARD_GRACE:-$(fm_poll_derived_grace "$POLL")}} +# Hard bound on a live holder's beacon age. Under it a re-arm refuses and asks +# for inspection (the grace above); at or past it the re-arm evicts the holder +# instead, because a watcher whose beacon has stalled that long is not polling +# and nothing else would ever replace it (evict_stalled_holder below). +WATCHER_STALL_BOUND=${FM_WATCHER_STALL_BOUND:-$((WATCHER_STALE_GRACE * 3))} HEARTBEAT=${FM_HEARTBEAT:-600} # base seconds between heartbeat scans HEARTBEAT_MAX=${FM_HEARTBEAT_MAX:-7200} # heartbeat backoff cap CHECK_INTERVAL=${FM_CHECK_INTERVAL:-300} # seconds between *.check.sh sweeps @@ -2324,12 +2335,40 @@ if ! fm_procevent_launch_confirm_seconds >/dev/null; then exit 1 fi -if ! fm_lock_try_acquire "$WATCH_LOCK"; then - BEAT="$STATE/.last-watcher-beat" +# evict_stalled_holder : retire a live lock holder whose beacon stalled past +# WATCHER_STALL_BOUND. The pid is signalled only while it still proves the +# lock's own recorded identity (fm_watcher_lock_matches_pid: this home, this +# script, and the starttime+cmdline proof the lock carries), so a recycled pid +# is never touched; TERM only, never KILL, and never a name or pattern match. +# Succeeds only once the holder has exited within the bounded wait. +evict_stalled_holder() { + local pid=$1 i=0 + fm_watcher_lock_matches_pid "$STATE" "$WATCH_PATH" "$pid" "$FM_HOME" || return 1 + kill -TERM "$pid" 2>/dev/null || return 1 + while [ "$i" -lt 50 ] && fm_pid_alive "$pid"; do + sleep 0.1 + i=$((i + 1)) + done + ! fm_pid_alive "$pid" +} + +EVICTED_PID= +EVICTED_BEAT_AGE= +BEAT="$STATE/.last-watcher-beat" +while ! fm_lock_try_acquire "$WATCH_LOCK"; do if [ -n "${FM_LOCK_HELD_PID:-}" ]; then if [ -e "$BEAT" ]; then beat_age=$(fm_path_age "$BEAT") if [ "$beat_age" -ge "$WATCHER_STALE_GRACE" ]; then + # One eviction per arm: the retry re-reads the lock and beacon, so a + # holder that exited leaves a dead-pid lock the normal reclaim takes, + # and a rival arm that won first reads as a fresh running watcher. + if [ -z "$EVICTED_PID" ] && [ "$beat_age" -ge "$WATCHER_STALL_BOUND" ] \ + && evict_stalled_holder "$FM_LOCK_HELD_PID"; then + EVICTED_PID=$FM_LOCK_HELD_PID + EVICTED_BEAT_AGE=$beat_age + continue + fi echo "watcher: lock held by live pid $FM_LOCK_HELD_PID but heartbeat is stale for ${beat_age}s (>${WATCHER_STALE_GRACE}s); inspect or stop that watcher before re-arming." >&2 exit 1 fi @@ -2342,6 +2381,9 @@ if ! fm_lock_try_acquire "$WATCH_LOCK"; then echo "watcher: already running" fi exit 0 +done +if [ -n "$EVICTED_PID" ]; then + echo "watcher: replaced stalled pid $EVICTED_PID (beacon ${EVICTED_BEAT_AGE}s past hard bound ${WATCHER_STALL_BOUND}s)" fi WATCHER_RECOVERY_PENDING=0 if [ -n "${FM_LOCK_RECOVERED_PID:-}" ]; then diff --git a/docs/configuration.md b/docs/configuration.md index e96a7b2c752..95acbe51ec3 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -2295,6 +2295,7 @@ FM_WATCH_REARM_RETRY_LIMIT=5 # Pi/OpenCode adapter launch-failure retries befo FM_WATCH_CYCLE_LOG_MAX_BYTES=262144 # size cap for the arm-owned watcher lifecycle ledger FM_WATCH_CYCLE_LOG_KEEP_LINES=1000 # newest complete lifecycle rows considered when the ledger is capped FM_WATCHER_STALE_GRACE=300 # defaults to FM_GUARD_GRACE if set, else the poll-derived grace (docs/turnend-guard.md "Guard grace and the poll cadence"); seconds a live watcher lock may have a stale beacon before re-arm errors +FM_WATCHER_STALL_BOUND= # defaults to 3x FM_WATCHER_STALE_GRACE; a live holder whose beacon is stale past this hard bound is evicted with TERM and replaced by the re-arm rather than refused (docs/turnend-guard.md, bin/fm-watch.sh header) FM_SIGNAL_GRACE=30 # seconds to coalesce nearby status and turn-end signals into one wake FM_TURNEND_CHURN_ABSORB_SECS=900 # longest one endpoint's bare turn-ends may be deferred on pane-churn evidence alone; only consulted when config/turnend-churn-absorb is present FM_CAPTAIN_RE='done:|needs-decision:|blocked:|failed:|PR ready|checks green|ready in branch|merged' # captain-relevant status regex; nonterminal progress verbs remain excluded even when their prose matches diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index bd293490d58..7328c502b2c 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -70,6 +70,7 @@ If `jq` is missing or hook stdin is empty, the guard exits 0 because it cannot s A fixed 300-second grace default stops correctly bounding staleness once a home's `FM_POLL` reaches or exceeds it: a perfectly healthy watcher mid-wait would then read stale at the edge of every full poll cycle by definition, which is exactly what a long-poll home (`FM_POLL=300`) hit against the Claude Stop-hook auto-arm (`bin/fm-claude-stop-autoarm.sh`). That hook and `bin/fm-watch.sh`'s own pre-acquisition staleness check (the "lock held by live pid but heartbeat is stale" refusal) both derive their default grace from the configured poll instead of a bare constant: `max(300, FM_POLL + 60)`, so the default never drops below the historical 300-second floor for the common short-poll case but grows with the poll cadence once that cadence would otherwise outrun it. `fm_poll_derived_grace` in `bin/fm-wake-lib.sh` is the single owner of that formula. +That refusal has a ceiling: once the live holder's beacon is stale past `FM_WATCHER_STALL_BOUND` (default three times the grace), the re-arm re-verifies the holder against the lock's recorded identity, retires it with TERM, and starts in its place, so a watcher wedged mid-cycle can no longer refuse every replacement indefinitely; `bin/fm-watch.sh`'s header owns the exact wording and the survives-TERM fallback. The auto-arm hook additionally exports its resolved `FM_GUARD_GRACE` when it forks `bin/fm-watch-arm.sh`, so the arm wrapper and the watcher it may start judge staleness with the exact same value the hook just judged it with, whether that value came from an operator override or the poll-derived default. `bin/fm-turnend-guard.sh`'s daemon-ownership branch (`fm_afk_daemon_owns_supervision`, above, covering both away and quiet mode) also derives its beacon grace from `fm_poll_derived_grace` rather than falling back to the bare 300-second default, for the same reason: the daemon's watcher-restart cadence there is not a fixed poll loop, so a flat grace misreads a daemon that is genuinely still cycling as down. Every other direct `FM_GUARD_GRACE` reader (`bin/fm-guard.sh`, the strict-watcher checks in `bin/fm-turnend-guard.sh` and its harness-specific wrappers, `bin/fm-wake-lib.sh`) still falls back to the bare 300-second default unless `FM_GUARD_GRACE` is set explicitly in the environment. diff --git a/tests/fm-watcher-lock.test.sh b/tests/fm-watcher-lock.test.sh index 78cacde3c82..315c5d3a2f5 100755 --- a/tests/fm-watcher-lock.test.sh +++ b/tests/fm-watcher-lock.test.sh @@ -172,6 +172,63 @@ test_live_stale_watch_lock_is_actionable() { pass "live watcher lock with stale heartbeat is actionable" } +test_live_stalled_watch_lock_is_replaced_past_hard_bound() { + # A live holder whose beacon is stale past the ordinary grace is refused, but + # a beacon stale past the hard bound evicts that holder (identity-verified + # TERM) and the arm starts in its place - the deadlock where every re-arm + # died against a live-but-stalled watcher while nothing polled the home. + local dir state fakebin out err status holder identity pid i lock_pid + dir=$(make_case live-stalled-lock) + state="$dir/state" + fakebin="$dir/fakebin" + out="$dir/watch.out" + err="$dir/watch.err" + sleep 300 & + holder=$! + identity=$(FM_STATE_OVERRIDE="$state" bash -c '. "$1"; fm_pid_identity "$2"' _ "$LIB" "$holder") || fail "could not identify the fake holder" + mkdir -p "$state/.watch.lock" + printf '%s\n' "$holder" > "$state/.watch.lock/pid" + printf '%s\n' "$dir" > "$state/.watch.lock/fm-home" + printf '%s\n' "$WATCH" > "$state/.watch.lock/watcher-path" + printf '%s\n' "$identity" > "$state/.watch.lock/pid-identity" + # Beacon decades old: past the grace, but a bound beyond it -> still refused. + touch -t 200001010000 "$state/.last-watcher-beat" + status=0 + PATH="$fakebin:$PATH" FM_HOME="$dir" FM_STATE_OVERRIDE="$state" FM_GUARD_GRACE=1 FM_WATCHER_STALL_BOUND=9999999999 FM_POLL=5 FM_SIGNAL_GRACE=1 FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" 2> "$err" || status=$? + [ "$status" -ne 0 ] || fail "watcher replaced a holder whose beacon was under the hard bound" + grep -F 'heartbeat is stale' "$err" >/dev/null || fail "under-bound stale holder lost its refusal" + is_live_non_zombie "$holder" || fail "under-bound stale holder was signalled" + # Same holder and beacon, a bound it is past -> evicted and replaced. + PATH="$fakebin:$PATH" FM_HOME="$dir" FM_STATE_OVERRIDE="$state" FM_GUARD_GRACE=1 FM_WATCHER_STALL_BOUND=3 FM_POLL=5 FM_SIGNAL_GRACE=1 FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" 2> "$err" & + pid=$! + i=0 + lock_pid= + while [ "$i" -lt 100 ]; do + lock_pid=$(cat "$state/.watch.lock/pid" 2>/dev/null || true) + [ "$lock_pid" = "$pid" ] && break + sleep 0.1 + i=$((i + 1)) + done + is_live_non_zombie "$pid" || fail "replacement watcher did not stay alive: $(cat "$err")" + [ "$lock_pid" = "$pid" ] || fail "replacement watcher did not take the lock (holder=$lock_pid)" + is_live_non_zombie "$holder" && fail "stalled holder survived the eviction" + # The lock pid is written inside fm_lock_try_acquire; the replacement message + # is echoed just after, so poll for the message rather than grep once and race + # the acquire/echo gap. + i=0 + while [ "$i" -lt 100 ]; do + grep -E "^watcher: replaced stalled pid $holder \(beacon [0-9]+s past hard bound 3s\)\$" "$out" >/dev/null && break + sleep 0.1 + i=$((i + 1)) + done + grep -E "^watcher: replaced stalled pid $holder \(beacon [0-9]+s past hard bound 3s\)\$" "$out" >/dev/null \ + || fail "watcher did not report the replacement: $(cat "$out" "$err")" + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + wait "$holder" 2>/dev/null || true + pass "live watcher lock with a beacon past the hard bound is replaced, under it is still refused" +} + test_guard_warnings() { # The guard's two operator-visible states, with resilient substrings instead of # four copy-coupled tests: @@ -1200,6 +1257,7 @@ test_msys_pid_identity_uses_proc test_stale_watch_lock_reclaimed test_stale_watch_reclaim_publishes_before_clear test_live_stale_watch_lock_is_actionable +test_live_stalled_watch_lock_is_replaced_past_hard_bound test_guard_warnings test_lock_single_winner_under_concurrency test_lock_steals_dead_pid_lock From 5cec4e2be2182c72b9b58feffc993b6fdacef534 Mon Sep 17 00:00:00 2001 From: Tiago Date: Fri, 25 Sep 2026 03:19:26 -0300 Subject: [PATCH 09/84] fix(pi): hide queued Firstmate inputs under Calm only when the session can keep them (#5563) * fix(pi): hide queued Firstmate notifications under Calm only when the session can keep them Calm now keeps authenticated Firstmate operational inputs out of Pi's queued-message listing, but only after proving the live session exposes every member needed to keep them across Escape. A session missing any of them keeps stock rows and Escape and shows one generic warning. Escape and the dequeue key return only captain-authored messages to the editor and re-queue hidden notifications in order; after an abort that kept any in Pi's agent queue, the adapter starts the delivery turn itself because Pi 0.87.1 does not continue an aborted run. Compaction-held notifications stay with Pi's compaction flush and never start or announce a turn. Fixes #1588 * docs(calm): record Pi 0.87.1 queued-row retention verification * no-mistakes(review): Deliver kept Calm notifications after tree-navigation aborts too * no-mistakes(review): Defer Calm notification turn until tree navigation finishes * no-mistakes(lint): Silence SC2016 for literal JavaScript in queue-retention e2e test --- .pi/extensions/fm-calm.ts | 14 +- .../lib/fm-calm-operational-user-layout.ts | 11 +- .../lib/fm-calm-pending-operational-layout.ts | 310 +++++++++++ .pi/extensions/lib/fm-operational-input.ts | 16 + bin/fm-test-run.sh | 1 + docs/calm-mode-feasibility.md | 50 ++ docs/calm.md | 14 +- tests/fm-calm-pi-extension.test.sh | 525 ++++++++++++++++++ ...m-calm-pi-queue-retention-live-e2e.test.sh | 153 +++++ tests/fm-pi-primary-live-e2e.test.sh | 1 + tests/fm-pi-primary-types.test.sh | 1 + 11 files changed, 1080 insertions(+), 16 deletions(-) create mode 100644 .pi/extensions/lib/fm-calm-pending-operational-layout.ts create mode 100755 tests/fm-calm-pi-queue-retention-live-e2e.test.sh diff --git a/.pi/extensions/fm-calm.ts b/.pi/extensions/fm-calm.ts index ec4a0380177..2db2af3af8c 100644 --- a/.pi/extensions/fm-calm.ts +++ b/.pi/extensions/fm-calm.ts @@ -6,10 +6,10 @@ // with a disposable component factory, and setHiddenThinkingLabel(). // ./lib/fm-calm-working-ship.ts owns the animated working presentation this file // installs. The focused tests pin those assumptions but never reject a -// newer Pi solely for its version. The collapsed-thinking and operational-user -// presentation adapters probe the exact API they patch and degrade independently with a -// diagnostic (see installCalmPresentationAdapter below) if a future Pi removes it; Pi -// still exposes no global renderer for arbitrary built-in or custom rows. +// newer Pi solely for its version. The collapsed-thinking, operational-user, and +// queued-operational presentation adapters probe the exact API they patch and degrade +// independently with a diagnostic (see installCalmPresentationAdapter below) if a future +// Pi removes it; Pi still exposes no global renderer for arbitrary built-in or custom rows. // docs/configuration.md owns the home-local Calm preference contract. // // Pi has one first-registration-wins ToolDefinition per tool name, with no merge or @@ -49,6 +49,10 @@ import { Box, Container, getKeybindings, type Component } from "@earendil-works/ import type { TSchema } from "typebox"; import { installCalmAssistantLayout } from "./lib/fm-calm-assistant-layout.ts"; import { installCalmOperationalUserLayout } from "./lib/fm-calm-operational-user-layout.ts"; +import { + installCalmPendingOperationalLayout, + refreshCalmPendingOperationalRows, +} from "./lib/fm-calm-pending-operational-layout.ts"; import { CALM_WORKING_SHIP_WIDGET_KEY, createCalmWorkingShipAnimation, @@ -122,6 +126,7 @@ function installCalmPresentationAdapter(name: string, install: () => void): void export default function (pi: ExtensionAPI) { installCalmPresentationAdapter("collapsed-thinking", installCalmAssistantLayout); installCalmPresentationAdapter("operational-user-row", installCalmOperationalUserLayout); + installCalmPresentationAdapter("queued-operational-row", installCalmPendingOperationalLayout); let exportRendering = false; let removeTerminalInputHandler: (() => void) | undefined; @@ -487,6 +492,7 @@ export default function (pi: ExtensionAPI) { // unchanged, which is what makes a toggle apply to rows already on screen. ctx.ui.setHiddenThinkingLabel(active ? "" : undefined); ctx.ui.setStatus("firstmate-calm", undefined); + refreshCalmPendingOperationalRows(); const expanded = ctx.ui.getToolsExpanded(); ctx.ui.setToolsExpanded(!expanded); diff --git a/.pi/extensions/lib/fm-calm-operational-user-layout.ts b/.pi/extensions/lib/fm-calm-operational-user-layout.ts index ca9b0bbcc0a..eb9fa374fac 100644 --- a/.pi/extensions/lib/fm-calm-operational-user-layout.ts +++ b/.pi/extensions/lib/fm-calm-operational-user-layout.ts @@ -6,7 +6,7 @@ import type { UserMessageComponent as PiUserMessageComponent } from "@earendil-works/pi-coding-agent"; import * as PiCodingAgent from "@earendil-works/pi-coding-agent"; import { calmPresentationHides } from "./fm-calm-visibility.ts"; -import { classifyFirstmateCurrentOperationalText } from "./fm-operational-input.ts"; +import { isFirstmateOperationalPresentationText } from "./fm-operational-input.ts"; type UserMessageConstructorArgs = ConstructorParameters; type UserMessageLike = { @@ -45,7 +45,6 @@ type CalmOperationalUserLayoutPatch = { const CALM_OPERATIONAL_USER_LAYOUT_PATCH = Symbol.for( "firstmate:calm-operational-user-layout:pi-0.81.1", ); -const LEGACY_CALM_OPERATIONAL_PREFIX = "\u2063Supervisor escalate ("; function contentIsTextOnly(content: unknown): boolean { if (typeof content === "string") return true; @@ -64,13 +63,7 @@ export function installCalmOperationalUserLayout(): void { [key: symbol]: CalmOperationalUserLayoutPatch | undefined; }; const hidesOperationalInput = (): boolean => calmPresentationHides("synthetic-user"); - const isOperationalInput = (text: string): boolean => { - if (!text.includes("\u2063")) return false; - return ( - classifyFirstmateCurrentOperationalText(text) !== undefined || - text.startsWith(LEGACY_CALM_OPERATIONAL_PREFIX) - ); - }; + const isOperationalInput = isFirstmateOperationalPresentationText; const installed = registry[CALM_OPERATIONAL_USER_LAYOUT_PATCH]; if (installed) { installed.hidesOperationalInput = hidesOperationalInput; diff --git a/.pi/extensions/lib/fm-calm-pending-operational-layout.ts b/.pi/extensions/lib/fm-calm-pending-operational-layout.ts new file mode 100644 index 00000000000..c9c2aeb7d15 --- /dev/null +++ b/.pi/extensions/lib/fm-calm-pending-operational-layout.ts @@ -0,0 +1,310 @@ +// Verified against Pi 0.87.1 (docs/calm-mode-feasibility.md), which draws queued +// "Steering:"/"Follow-up:" rows, their spacer, and the dequeue hint in +// InteractiveMode.updatePendingMessagesDisplay from InteractiveMode.getAllQueuedMessages. +// A Firstmate notification sent while a turn runs waits there before it is ever a chat row, +// so ./fm-calm-operational-user-layout.ts never sees it. This adapter filters only what that +// one listing reads; the queue Pi delivers from and persists is untouched. +// +// Hiding a queued row makes Pi's InteractiveMode.restoreQueuedMessagesToEditor (Escape during +// a run, and the dequeue key) the one place hidden text could come back: stock Pi empties the +// whole queue into the editor through clearAllQueues. Two rules are absolute: a notification +// this adapter hid never reappears as raw text, and none is dropped to keep presentation +// clean. Under Calm the restore hands only the other messages to the editor and puts the +// hidden notifications back in the queue in their original order. +// +// Putting them back needs members that live on the session object rather than the +// prototype, so they cannot be probed at install. Each session is checked on its first +// queued-listing draw while Calm is on, before any row is hidden. A session missing any of them +// gets no queued-row hiding at all and one warning; its rows and Escape stay stock. +// See https://github.com/kunchenguid/firstmate/issues/1588. +// +// Pi 0.87.1 stops its run loop once a restore is followed by an abort (Escape, or navigating +// the session tree during a run), so a queue that still holds messages when the aborted run +// settles is not delivered until something else starts a turn. After any restore that kept +// notifications in Pi's agent queue, this adapter waits for the session to settle and, if it +// is idle with messages still queued, starts that turn itself with one generic status line. +// A run that keeps going drains the queue itself, so nothing starts after a plain dequeue. A +// notification kept only in the compaction queue is flushed by Pi when compaction ends, so +// it neither counts toward that turn nor announces one. +import * as PiCodingAgent from "@earendil-works/pi-coding-agent"; +import { calmPresentationHides } from "./fm-calm-visibility.ts"; +import { isFirstmateOperationalPresentationText } from "./fm-operational-input.ts"; + +type QueuedMessages = { + steering: string[]; + followUp: string[]; +}; +type CompactionQueuedMessage = { + text: string; + mode: string; +}; +type RetainingSession = { + getSteeringMessages(): readonly string[]; + getFollowUpMessages(): readonly string[]; + clearQueue(): QueuedMessages; + _queueSteer(text: string): unknown; + _queueFollowUp(text: string): unknown; + waitForIdle(): Promise; + sendUserMessage(content: string): Promise; + readonly isIdle: boolean; +}; +type PendingRowsHost = { + session: unknown; + compactionQueuedMessages: CompactionQueuedMessage[]; + showStatus?(message: string): void; + showWarning?(message: string): void; + updatePendingMessagesDisplay(): void; +}; +type RestoreOptions = { + abort?: boolean; + currentText?: string; +}; +type InteractiveModePendingPrototype = { + getAllQueuedMessages(this: PendingRowsHost): QueuedMessages; + updatePendingMessagesDisplay(this: PendingRowsHost): void; + clearAllQueues(this: PendingRowsHost): QueuedMessages; + restoreQueuedMessagesToEditor(this: PendingRowsHost, options?: RestoreOptions): number; +}; +type CalmPendingOperationalLayoutPatch = { + hidesOperationalInput: () => boolean; + isOperationalInput: (text: string) => boolean; + refresh: () => void; +}; +type Restoring = { + session: RetainingSession; + retains: (text: string) => boolean; + keptInAgentQueue: number; +}; + +export const CALM_QUEUE_RETENTION_SESSION_METHODS = [ + "getSteeringMessages", + "getFollowUpMessages", + "clearQueue", + "_queueSteer", + "_queueFollowUp", + "waitForIdle", + "sendUserMessage", +] as const; + +// Generic by design: no notification text, marker, kind, path, or identifier. +export const CALM_QUEUED_ROWS_UNSUPPORTED_WARNING = + "Firstmate Calm: this Pi session cannot keep queued messages across Escape, so queued Firstmate rows stay visible."; +export const CALM_SUPERVISION_CONTINUES_NOTICE = + "Firstmate supervision continues in a new turn."; + +// Keep the introduction-version symbol stable so a compatible upgrade cannot +// double-patch a live process. +const CALM_PENDING_OPERATIONAL_LAYOUT_PATCH = Symbol.for( + "firstmate:calm-pending-operational-layout:pi-0.87.1", +); + +function settle(queued: unknown): void { + void Promise.resolve(queued).catch(() => {}); +} + +export function installCalmPendingOperationalLayout(): void { + const registry = globalThis as typeof globalThis & { + [key: symbol]: CalmPendingOperationalLayoutPatch | undefined; + }; + const hidesOperationalInput = (): boolean => calmPresentationHides("synthetic-user"); + const installed = registry[CALM_PENDING_OPERATIONAL_LAYOUT_PATCH]; + if (installed) { + installed.hidesOperationalInput = hidesOperationalInput; + installed.isOperationalInput = isFirstmateOperationalPresentationText; + return; + } + + const InteractiveMode = PiCodingAgent.InteractiveMode; + if (typeof InteractiveMode !== "function") { + throw new Error("Firstmate Calm requires Pi InteractiveMode"); + } + const prototype = InteractiveMode.prototype as unknown as InteractiveModePendingPrototype; + const originalGetAllQueuedMessages = prototype.getAllQueuedMessages; + const originalUpdatePendingMessagesDisplay = prototype.updatePendingMessagesDisplay; + const originalClearAllQueues = prototype.clearAllQueues; + const originalRestoreQueuedMessagesToEditor = prototype.restoreQueuedMessagesToEditor; + for (const [name, method] of [ + ["getAllQueuedMessages", originalGetAllQueuedMessages], + ["updatePendingMessagesDisplay", originalUpdatePendingMessagesDisplay], + ["clearAllQueues", originalClearAllQueues], + ["restoreQueuedMessagesToEditor", originalRestoreQueuedMessagesToEditor], + ] as const) { + if (typeof method !== "function") { + throw new Error(`Firstmate Calm requires Pi InteractiveMode.${name}`); + } + } + + // The interactive mode that last drew queued rows, so a /calm toggle can redraw them. + let lastHost: PendingRowsHost | undefined; + const patch: CalmPendingOperationalLayoutPatch = { + hidesOperationalInput, + isOperationalInput: isFirstmateOperationalPresentationText, + refresh: () => lastHost?.updatePendingMessagesDisplay(), + }; + + const retentionBySession = new WeakMap(); + function retainingSession(host: PendingRowsHost): RetainingSession | undefined { + const session = host.session; + if (typeof session !== "object" || session === null) return undefined; + let supported = retentionBySession.get(session); + if (supported === undefined) { + const members = session as Record; + supported = + CALM_QUEUE_RETENTION_SESSION_METHODS.every((name) => typeof members[name] === "function") && + typeof members.isIdle === "boolean" && + Array.isArray(host.compactionQueuedMessages); + retentionBySession.set(session, supported); + if (!supported) { + if (typeof host.showWarning === "function") { + host.showWarning(CALM_QUEUED_ROWS_UNSUPPORTED_WARNING); + } else { + console.error(CALM_QUEUED_ROWS_UNSUPPORTED_WARNING); + } + } + } + return supported ? (session as RetainingSession) : undefined; + } + + // What the latest draw of the queued listing actually hid, and for which session. The + // restore retains from this record rather than a fresh classification, so a row the + // captain never saw stays hidden even if the classifier cannot answer a second time. + let hidden: { session: object; texts: Set } | undefined; + // Set only for the synchronous draw below, so every other reader of the queue still + // sees exactly what Pi queued. + let hidingInto: Set | undefined; + // Set only for the synchronous restore below, so any other clearAllQueues caller keeps + // Pi's stock semantics. + let restoring: Restoring | undefined; + + prototype.getAllQueuedMessages = function (this: PendingRowsHost): QueuedMessages { + const queued = originalGetAllQueuedMessages.call(this); + const texts = hidingInto; + if (!texts) return queued; + const stays = (text: string): boolean => { + if (!patch.isOperationalInput(text)) return true; + texts.add(text); + return false; + }; + return { + ...queued, + steering: queued.steering.filter(stays), + followUp: queued.followUp.filter(stays), + }; + }; + + prototype.updatePendingMessagesDisplay = function (this: PendingRowsHost): void { + lastHost = this; + if (!patch.hidesOperationalInput() || !retainingSession(this)) { + hidden = undefined; + originalUpdatePendingMessagesDisplay.call(this); + return; + } + const texts = new Set(); + hidingInto = texts; + try { + // Pi skips the spacer and dequeue hint when nothing is left to list, so an + // all-operational queue draws no rows at all. + originalUpdatePendingMessagesDisplay.call(this); + } finally { + hidingInto = undefined; + } + hidden = texts.size > 0 ? { session: this.session as object, texts } : undefined; + }; + + prototype.clearAllQueues = function (this: PendingRowsHost): QueuedMessages { + const current = restoring; + if (!current) return originalClearAllQueues.call(this); + const { session, retains } = current; + const steering = session.getSteeringMessages().filter(retains); + const followUp = session.getFollowUpMessages().filter(retains); + const compaction = this.compactionQueuedMessages.filter((message) => retains(message.text)); + const cleared = originalClearAllQueues.call(this); + if (steering.length + followUp.length + compaction.length === 0) return cleared; + // Pi's already-expanded queueing entry points: no input handler or template expansion + // runs a second time on text that already went through them once. + for (const text of steering) settle(session._queueSteer(text)); + for (const text of followUp) settle(session._queueFollowUp(text)); + this.compactionQueuedMessages.push(...compaction); + current.keptInAgentQueue = steering.length + followUp.length; + return { + ...cleared, + steering: cleared.steering.filter((text) => !retains(text)), + followUp: cleared.followUp.filter((text) => !retains(text)), + }; + }; + + prototype.restoreQueuedMessagesToEditor = function ( + this: PendingRowsHost, + options?: RestoreOptions, + ): number { + const hidesNow = patch.hidesOperationalInput(); + const hiddenTexts = hidden && hidden.session === this.session ? hidden.texts : undefined; + const session = hidesNow || hiddenTexts ? retainingSession(this) : undefined; + if (!session) return originalRestoreQueuedMessagesToEditor.call(this, options); + + // A notification queued since the last draw was never shown either, so while Calm + // hides, it is kept the same way; classification is asked once per text. + const answers = new Map(); + const retains = (text: string): boolean => { + if (hiddenTexts?.has(text)) return true; + if (!hidesNow) return false; + let answer = answers.get(text); + if (answer === undefined) { + answer = patch.isOperationalInput(text); + answers.set(text, answer); + } + return answer; + }; + const current: Restoring = { session, retains, keptInAgentQueue: 0 }; + restoring = current; + try { + return originalRestoreQueuedMessagesToEditor.call(this, options); + } finally { + restoring = undefined; + if (current.keptInAgentQueue > 0) continueWhenSettled(this, session); + } + }; + + // Delivers what a settled run left queued. Messages already in the queue cannot start a + // turn by themselves, so the first is taken out and sent as the turn's prompt and the rest + // are put back behind it: steering first, then follow-ups, the order Pi delivers them in. + function continueWhenSettled(host: PendingRowsHost, session: RetainingSession): void { + const settled = async (): Promise => { + do { + await session.waitForIdle(); + // Pi resolves idle waiters in microtasks, and tree navigation resumes from its + // abort in the same microtask run and marks the session busy before its first await. + // Yielding a macrotask lets that navigation claim the session, so the turn starts on + // the navigated branch instead of racing it on the abandoned one. + await new Promise((resolve) => setTimeout(resolve, 0)); + if (host.session !== session) return false; + } while (!session.isIdle); + return true; + }; + settled() + .then((idle) => { + if (!idle) return; + const { steering, followUp } = session.clearQueue(); + const first = steering.length > 0 ? steering.shift() : followUp.shift(); + if (first === undefined) return; + for (const text of steering) settle(session._queueSteer(text)); + for (const text of followUp) settle(session._queueFollowUp(text)); + host.showStatus?.(CALM_SUPERVISION_CONTINUES_NOTICE); + // Pi rejects before recording the prompt when it cannot start the turn, so the + // message is queued again rather than lost. + session.sendUserMessage(first).catch(() => settle(session._queueFollowUp(first))); + }) + .catch(() => {}); + } + + registry[CALM_PENDING_OPERATIONAL_LAYOUT_PATCH] = patch; +} + +// Redraws the queued listing after a /calm toggle so rows already listed follow the new +// choice at once instead of at the next queue change. +export function refreshCalmPendingOperationalRows(): void { + const registry = globalThis as typeof globalThis & { + [key: symbol]: CalmPendingOperationalLayoutPatch | undefined; + }; + registry[CALM_PENDING_OPERATIONAL_LAYOUT_PATCH]?.refresh(); +} diff --git a/.pi/extensions/lib/fm-operational-input.ts b/.pi/extensions/lib/fm-operational-input.ts index 4070684c6a4..e697e383fec 100644 --- a/.pi/extensions/lib/fm-operational-input.ts +++ b/.pi/extensions/lib/fm-operational-input.ts @@ -121,3 +121,19 @@ export function classifyFirstmateCurrentOperationalText( ): string | undefined { return runOperationalInputCommand("kind", content); } + +// The only legacy operational shape Calm presentation hides on top of the current +// typed kinds. The broader `classify` legacy set stays out: its bare forms are text a +// captain can type, so hiding them would hide real input. +const LEGACY_CALM_OPERATIONAL_PREFIX = "\u2063Supervisor escalate ("; + +// Single owner of "may Calm presentation hide this exact input?", shared by the +// transcript-row and queued-row adapters so the two can never disagree about a message. +// Text without the U+2063 marker answers here without spawning the classifier. +export function isFirstmateOperationalPresentationText(text: string): boolean { + if (!text.includes("\u2063")) return false; + return ( + classifyFirstmateCurrentOperationalText(text) !== undefined || + text.startsWith(LEGACY_CALM_OPERATIONAL_PREFIX) + ); +} diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 30cd64f56eb..d460c3ffe6c 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -365,6 +365,7 @@ family_for_basename() { fm-quota-array-dispatch-live-e2e.test.sh|fm-send-secondmate-marker-herdr-e2e.test.sh|\ fm-send-inbox-doorbell-live-e2e.test.sh|\ fm-calm-claude-mod-plugin.test.sh|fm-calm-claude-mod-live-e2e.test.sh|\ + fm-calm-pi-queue-retention-live-e2e.test.sh|\ fm-herdr-submit-confirm-live-e2e.test.sh) printf '%s\n' live-harness-optin ;; diff --git a/docs/calm-mode-feasibility.md b/docs/calm-mode-feasibility.md index 128f9435945..0136b849dc2 100644 --- a/docs/calm-mode-feasibility.md +++ b/docs/calm-mode-feasibility.md @@ -279,6 +279,24 @@ For the duplicate-turn fix and the latest presentation change, the launch templa The canonical encoder and every non-Pi delivery path remain unchanged, and the tmux, Herdr, Zellij, Orca, and cmux runtime surfaces continue to transport the same input selected by the harness adapter. Pi's Calm implementation changed only to consume the shared sprite core, while the new Claude Code mod changes drawings only; every producer and non-Pi transport remains unchanged. +## Queued operational-row retention + +On Pi 0.87.1 with Calm persisted on, a Firstmate watcher notification sent while a tool held the turn was listed under the running turn as `Follow-up: FIRSTMATE_OP: v1 watcher: ...`, identical to Calm off. +Pressing Escape moved that raw text into the editor and removed it from Pi's queue, and the session recorded no delivery of it, so a captain who cleared the editor lost the notification. +The initiating trigger was a notification queued during a run. +The exposure condition was that Pi draws queued input in `InteractiveMode.updatePendingMessagesDisplay` and restores it through `restoreQueuedMessagesToEditor`, a path separate from the `addMessageToChat` path the operational-user adapter covers. +The visible symptom was the listed row and, after Escape, the raw text in the editor. + +Hiding the listed row alone would turn the Escape path into the defect issue #1588 describes: stock restore joins the whole queue into the editor, so a hidden notification would reappear as raw text. +Keeping it queued across the restore needs the session's already-expanded queueing entry points (`_queueSteer` and `_queueFollowUp`) and, for the delivery below, `clearQueue`, `waitForIdle`, `sendUserMessage`, and `isIdle`. +Those live on the session instance reached through `InteractiveMode.session`, so they are checked per session before the first row is hidden rather than at extension load. + +A counterfactual built from the closed PR #1620 adapter hid the row and kept the notification out of the editor, but Pi 0.87.1's `AgentSession._runAgentPrompt` stops continuing once an abort was requested, so the kept follow-up stayed queued until the captain's next prompt while the adapter announced a new turn. +The shipped adapter therefore starts that turn itself once the aborted run settles: it takes the first queued message out, sends it with `sendUserMessage`, and puts the rest back behind it in Pi's delivery order. +Navigating the session tree during a run takes the same path without an abort flag, restoring the queue and then calling `session.abort()`, so the adapter waits for every restore that kept a notification and starts the turn only if the session is then idle with messages still queued. +Pi starts `navigateTree` in the same microtask run that resumes from that abort and marks the session busy before its first await, so the adapter yields one macrotask after each idle wait and waits again while the session is busy, which starts the turn on the navigated branch instead of racing the navigation on the abandoned one. +The same real-Pi reproduction then delivered the notification exactly once in a new turn, returned a queued captain message to the editor, and left Calm off stock. + ## Regression coverage `tests/fm-calm-pi-extension.test.sh` compares wrapped and stock renderers and verifies all seven built-ins plus `fm_watch_arm_pi`; `tests/fm-pi-branch-extension.test.sh` verifies `fm_branch_outcomes` Calm toggling, capability-probed all-line versus collapsed stock output, exact expanded output, and export rendering. @@ -288,6 +306,8 @@ A native deterministic `/skill:ahoy` turn produces thinking, tool-call, and tool The operational provider path covers Calm loaded on, loaded off, default preference, extension absent, exact watcher delivery, narrow bare-marker legacy input, persisted restart replay, a genuine captain prompt, and adjacent notifications coalesced into one intended processing turn. It asserts one persisted and rendered captain answer, exact user-role operational envelopes in order, no replacement custom messages, one processing result, zero operational transcript rows, and the two-row neighboring-assistant geometry for live, adjacent, and restart paths. Quoted current markers, ASCII-only labels, ordinary text before a marker, unrelated U+2063 placement, and image-bearing input remain visible in component and native transcript checks. +Queued-row coverage drives Pi's real listing and restore methods over a stand-in session for each capability-check branch, including a hidden row kept when the classifier cannot answer again and a refused continuation that re-queues instead of dropping, and repeats Escape in a real Pi TUI with Calm on, with a captain message queued beside the notification, and with Calm off. +`tests/fm-calm-pi-queue-retention-live-e2e.test.sh` is the default-on, token-free guard that probes a running Pi session for every member the check requires and fails naming the installed Pi version. `tests/fm-pi-primary-live-e2e.test.sh` also proves the working ship replaces the built-in `Working...` row while Calm is active on the credentialed provider path, and that it clears when the run settles, before continuing its ordinary watcher lifecycle. `tests/fm-pi-primary-types.test.sh` performs strict no-emit TypeScript checking against whichever Pi declarations are installed, without pinning a version of its own. `tests/fm-calm-claude-mod.test.sh` needs no Claude Code binary: it proves the mod is one hooks module with no command, skill, agent, or classic hook path around its opt-in, that Pi's working ship renders byte-for-byte the shared sprite core painted in ANSI at every width and step, that the Raster packing lays that frame out exactly, that the mod resolves its home like Pi, that its live and restored working-note classifiers enforce the visibility boundaries [`calm.md`](calm.md#claude-code) owns, and that its operational-input classifier agrees with `bin/fm-operational-input.sh` on a corpus the shell owner itself encodes plus legacy shapes and near misses. @@ -301,6 +321,7 @@ tests/fm-calm-pi-extension.test.sh tests/fm-pi-branch-extension.test.sh FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh tests/fm-pi-primary-types.test.sh +tests/fm-calm-pi-queue-retention-live-e2e.test.sh tests/fm-calm-claude-mod.test.sh tests/fm-calm-claude-mod-plugin.test.sh FM_CLAUDE_CALM_LIVE_E2E=1 tests/fm-calm-claude-mod-live-e2e.test.sh @@ -621,6 +642,35 @@ ok - the rendered-export-DOM guard renders in one pass, retries a bounded number ok - Pi calm native E2E replaces the stock working row with a moving, resize-clamped working ship that freezes and resumes across two working periods in one Pi session, clears on abort, keeps captain turns visible, hides exact operational user rows without changing persistence, restores stock rendering Calm-off, survives restart, and preserves export plus Ctrl+O behavior ``` +## 2026-09-24 Pi 0.87.1 queued-row retention verification + +The queued-row adapter was verified on Linux 7.0.0 x86_64, Node v22.23.1, and tmux against the globally installed `@earendil-works/pi-coding-agent` 0.87.1, with TypeScript 7.0.2 installed only for the typecheck. +Every Pi run used a scratch home, project, agent directory, and session directory with a local faux provider, so no model request left the machine. + +```sh +pi --version +tests/fm-calm-pi-queue-retention-live-e2e.test.sh +tests/fm-calm-pi-extension.test.sh +tests/fm-pi-primary-types.test.sh +``` + +```text +0.87.1 +ok - Pi 0.87.1 exposes every queue-retention member Calm preflights before hiding queued Firstmate rows +ok - Calm hides queued Firstmate rows only on a session that can keep them, keeps hidden ones out of the editor on Escape, delivers them once in order, and leaves unsupported sessions and Calm off stock +ok - Pi 0.87.1 with Calm on keeps a queued Firstmate notification unlisted, out of the editor on Escape, and delivers it once in a new announced turn, while Calm off stays stock +ok - tracked Pi extensions pass strict no-emit typecheck against Pi 0.87.1 +``` + +The rest of `tests/fm-calm-pi-extension.test.sh` passed unchanged in the same run. +With a member name the running session does not have added to the adapter's required list, the live guard failed as designed: + +```text +not ok - Pi 0.87.1 lacks the queue-retention capability Calm needs to hide queued Firstmate rows: session._queueNotARealMember +``` + +With the queued-row adapter left uninstalled, the real-Pi Escape case failed on the listed notification, `Pi Calm listed a queued Firstmate notification`. + ## 2026-09-15 Claude Code 2.1.272 mods feasibility and the shipped mod Claude Code 2.1.272 exposes exactly the capability the 2026-07-22 row found missing, through its early-access "Claude Mods" surface, whose engineering primitive is the function hook: a plugin whose behavior lives in one hooks module exporting `register(on, options)`, hooking dotted engine events as `($, e, next)` middleware, with `ui.render` drawing per-component transcript rows and the working row, `$.ui.invalidate("ui.render")` redrawing every hooked drawing, and `$.ui.blit` repainting a mounted `Raster` without a render pass. diff --git a/docs/calm.md b/docs/calm.md index f590027df40..52745ec9909 100644 --- a/docs/calm.md +++ b/docs/calm.md @@ -24,6 +24,10 @@ Pi applies that rule independently to each text block, so a short working note c A working note is briefly visible while it streams before its settled row collapses. The narration is hidden only from the live transcript presentation, and remains in the message, model context, session storage, and `/export` artifacts. The operational inputs Calm classifies remain ordinary user-role messages, while Pi's transcript layout renders their complete rows at zero height. +While a turn runs, Calm also keeps those Firstmate inputs out of Pi's queued-message listing, and the captain's own queued messages stay listed. +Escape and the dequeue key return only the captain's queued messages to the editor; hidden Firstmate inputs stay queued in their original order and are never shown as raw text or dropped. +When Escape, or navigating the session tree, stops a run with Firstmate inputs still queued, Calm starts one new turn to deliver them and shows the one-line notice `Firstmate supervision continues in a new turn.` +Inputs held behind a running compaction stay there until Pi sends them after compaction, so they start and announce no turn of their own. The session-start nudge remains on its existing non-displayed custom-message path. Outside Pi's same-name built-in override collision described below, Calm changes presentation only. @@ -39,8 +43,11 @@ These are supported-API boundaries rather than hidden-content failures. ## Pi compatibility Calm has no numeric Pi version minimum or maximum and never refuses Pi solely because its version is newer than a previously verified version. -The collapsed-thinking and operational-user-row presentation adapters probe the exact Pi API seam they patch when Calm loads. -If Pi removes one of those seams, Calm logs a diagnostic naming the unavailable adapter and skips only that adapter; `/calm`, the other adapter, and unrelated Pi extensions remain available. +The collapsed-thinking, operational-user-row, and queued-operational-row presentation adapters probe the exact Pi API seam they patch when Calm loads. +If Pi removes one of those seams, Calm logs a diagnostic naming the unavailable adapter and skips only that adapter; `/calm`, the other adapters, and unrelated Pi extensions remain available. +Keeping hidden queued inputs across Escape also needs members of Pi's live session, which exist only once a session runs. +Calm checks them for each session on its first queued-listing draw, before hiding anything. +A session missing any of them keeps its queued rows and Escape exactly as stock and shows one warning, and `tests/fm-calm-pi-queue-retention-live-e2e.test.sh` fails naming the installed Pi version. Calm's built-in tool presentation (`bash`, `read`, `edit`, `write`, `grep`, `find`, `ls`) shares Pi's single, unmerged override slot per name with any other extension that overrides the same tool. While the persisted Calm preference is off, Calm registers none of those overrides and therefore contests no built-in tool name. @@ -52,7 +59,7 @@ If the other extension wins, a session-start console diagnostic names the tool a [`calm-mode-feasibility.md`](calm-mode-feasibility.md) owns the version-scoped renderer taxonomy, built-in override constraints, and empirical evidence. [`configuration.md`](configuration.md#calm-preference-configcalm) owns the persisted preference file and resolution rules. -`.pi/extensions/lib/fm-calm-visibility.ts` owns the visibility policy, `.claude/mods/firstmate-calm/lib/fm-calm-preservation.ts` owns the shared substantive mid-turn text rule that Pi imports through its tracked symlink, `.pi/extensions/lib/fm-calm-operational-user-layout.ts` owns the zero-height operational-user row adapter, and `.pi/extensions/lib/fm-calm-working-ship.ts` owns Pi's animated working presentation over the sprite geometry both harnesses share in `.claude/mods/firstmate-calm/lib/fm-calm-working-ship-sprite.ts`. +`.pi/extensions/lib/fm-calm-visibility.ts` owns the visibility policy, `.claude/mods/firstmate-calm/lib/fm-calm-preservation.ts` owns the shared substantive mid-turn text rule that Pi imports through its tracked symlink, `.pi/extensions/lib/fm-calm-operational-user-layout.ts` owns the zero-height operational-user row adapter, `.pi/extensions/lib/fm-calm-pending-operational-layout.ts` owns the queued-row adapter and its session capability check, and `.pi/extensions/lib/fm-calm-working-ship.ts` owns Pi's animated working presentation over the sprite geometry both harnesses share in `.claude/mods/firstmate-calm/lib/fm-calm-working-ship-sprite.ts`. Regression entry points: @@ -60,6 +67,7 @@ Regression entry points: tests/fm-calm-pi-extension.test.sh tests/fm-pi-branch-extension.test.sh tests/fm-pi-primary-types.test.sh +tests/fm-calm-pi-queue-retention-live-e2e.test.sh FM_PI_LIVE_E2E=1 tests/fm-pi-primary-live-e2e.test.sh ``` diff --git a/tests/fm-calm-pi-extension.test.sh b/tests/fm-calm-pi-extension.test.sh index bf9a967e204..5fde98777da 100755 --- a/tests/fm-calm-pi-extension.test.sh +++ b/tests/fm-calm-pi-extension.test.sh @@ -10,6 +10,7 @@ EXT="$ROOT/.pi/extensions/fm-calm.ts" ASSISTANT_LAYOUT="$ROOT/.pi/extensions/lib/fm-calm-assistant-layout.ts" PRESERVATION="$ROOT/.pi/extensions/lib/fm-calm-preservation.ts" OPERATIONAL_USER_LAYOUT="$ROOT/.pi/extensions/lib/fm-calm-operational-user-layout.ts" +PENDING_OPERATIONAL_LAYOUT="$ROOT/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" VISIBILITY="$ROOT/.pi/extensions/lib/fm-calm-visibility.ts" WORKING_SHIP="$ROOT/.pi/extensions/lib/fm-calm-working-ship.ts" WORKING_SHIP_SPRITE="$ROOT/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" @@ -190,6 +191,7 @@ test_home_resolution() { cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" cp "$PRESERVATION" "$fixture/project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$PENDING_OPERATIONAL_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship.ts" cp "$WORKING_SHIP_SPRITE" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" @@ -314,6 +316,7 @@ test_pi_compat_degraded_adapter() { cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" cp "$PRESERVATION" "$fixture/project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$PENDING_OPERATIONAL_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship.ts" cp "$WORKING_SHIP_SPRITE" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" @@ -415,6 +418,7 @@ test_pi_compat_missing_adapter_exports() { cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" cp "$PRESERVATION" "$fixture/project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$PENDING_OPERATIONAL_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship.ts" cp "$WORKING_SHIP_SPRITE" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" @@ -431,10 +435,12 @@ test_pi_compat_missing_adapter_exports() { out=$(cd "$fixture/project" && node --input-type=module 2>&1 <<'JS' const assistant = await import("./.pi/extensions/lib/fm-calm-assistant-layout.ts"); const operational = await import("./.pi/extensions/lib/fm-calm-operational-user-layout.ts"); +const pending = await import("./.pi/extensions/lib/fm-calm-pending-operational-layout.ts"); for (const [name, install, expected] of [ ["collapsed-thinking", assistant.installCalmAssistantLayout, "AssistantMessageComponent"], ["operational-user-row", operational.installCalmOperationalUserLayout, "InteractiveMode"], + ["queued-operational-row", pending.installCalmPendingOperationalLayout, "InteractiveMode"], ]) { let reason; try { @@ -456,6 +462,319 @@ JS pass "missing Pi presentation class exports reach the independent adapter degradation path" } +# Pi draws queued input in its own listing, and Escape empties that queue into the editor. +# This drives Pi's real listing and restore methods over a stand-in session so every +# branch of the queue-retention preflight is pinned without a harness; the tmux case in +# test_queued_operational_escape_e2e covers the same path in a real Pi. +test_queued_operational_rows() { + local fixture out status + if ! command -v node >/dev/null 2>&1 || ! command -v npm >/dev/null 2>&1; then + echo "skip: node or npm not found for Pi Calm queued-row test" + return 0 + fi + if [ ! -f "$PI_PACKAGE_DIR/package.json" ]; then + echo "skip: installed @earendil-works/pi-coding-agent package not found" + return 0 + fi + + fixture="$TMP_ROOT/queued-operational-rows" + mkdir -p "$fixture/lib" "$fixture/node_modules/@earendil-works" + cp "$PENDING_OPERATIONAL_LAYOUT" "$fixture/lib/fm-calm-pending-operational-layout.ts" + cp "$VISIBILITY" "$fixture/lib/fm-calm-visibility.ts" + cp "$PI_OPERATIONAL_INPUT" "$fixture/lib/fm-operational-input.ts" + ln -s "$PI_PACKAGE_DIR" "$fixture/node_modules/@earendil-works/pi-coding-agent" + ln -s "$PI_PACKAGE_DIR/node_modules/@earendil-works/pi-tui" "$fixture/node_modules/@earendil-works/pi-tui" + ln -s "$PI_PACKAGE_DIR/node_modules/typebox" "$fixture/node_modules/typebox" + printf '%s\n' '{"type":"module"}' >"$fixture/package.json" + # Lets the fixture take the classifier away mid-run, the way a missing or broken + # bin/fm-operational-input.sh would. + cat >"$fixture/operational-input-probe.sh" <<'SH' +#!/usr/bin/env bash +[ -e "$FM_CLASSIFIER_DOWN" ] && exit 3 +exec "$FM_OPERATIONAL_INPUT_OWNER" "$@" +SH + chmod +x "$fixture/operational-input-probe.sh" + + out=$(cd "$fixture" && \ + FM_OPERATIONAL_INPUT_SCRIPT="$fixture/operational-input-probe.sh" \ + FM_OPERATIONAL_INPUT_OWNER="$OPERATIONAL_INPUT" \ + FM_CLASSIFIER_DOWN="$fixture/classifier-down" \ + PI_PACKAGE_DIR="$PI_PACKAGE_DIR" \ + node --input-type=module 2>&1 <<'JS' +import { rmSync, writeFileSync } from "node:fs"; +import { pathToFileURL } from "node:url"; + +const packageRoot = process.env.PI_PACKAGE_DIR; +const [{ InteractiveMode }, { initTheme }] = await Promise.all([ + import(pathToFileURL(`${packageRoot}/dist/modes/interactive/interactive-mode.js`).href), + import(pathToFileURL(`${packageRoot}/dist/modes/interactive/theme/theme.js`).href), +]); +initTheme("dark"); +const layout = await import("./lib/fm-calm-pending-operational-layout.ts"); +const visibility = await import("./lib/fm-calm-visibility.ts"); +const operationalInput = await import("./lib/fm-operational-input.ts"); +layout.installCalmPendingOperationalLayout(); + +const check = (condition, message) => { + if (!condition) throw new Error(message); +}; +const settle = () => new Promise((resolve) => setTimeout(resolve, 20)); +const stripAnsi = (text) => text.replace(/\x1b\[[0-9;]*m/g, ""); +const watcherOne = operationalInput.encodeFirstmateOperationalInput("watcher", "QUEUED_MONITOR_ONE"); +const watcherTwo = operationalInput.encodeFirstmateOperationalInput("watcher", "QUEUED_MONITOR_TWO"); +const legacyAway = "⁣Supervisor escalate (QUEUED_LEGACY_AWAY)"; +const captainText = "CAPTAIN_QUEUED_TEXT"; +// A captain can type the marker's words; only the authenticated envelope may hide. +const lookalike = "FIRSTMATE_OP: v1 away-supervisor: CAPTAIN_TYPED_LOOKALIKE"; +const operationalTexts = [watcherOne, watcherTwo, legacyAway]; + +// Implements every session member the retention relies on with Pi's own semantics: +// queueing appends, clearQueue empties both lists, and a prompt only starts when idle. +function makeSession({ missing = [], rejectPrompt = false } = {}) { + const session = { + steering: [], + followUp: [], + prompts: [], + aborts: 0, + idle: false, + idleWaiters: [], + getSteeringMessages() { return this.steering; }, + getFollowUpMessages() { return this.followUp; }, + clearQueue() { + const cleared = { steering: [...this.steering], followUp: [...this.followUp] }; + this.steering = []; + this.followUp = []; + return cleared; + }, + async _queueSteer(text) { this.steering.push(text); }, + async _queueFollowUp(text) { this.followUp.push(text); }, + waitForIdle() { + return this.idle ? Promise.resolve() : new Promise((resolve) => this.idleWaiters.push(resolve)); + }, + async sendUserMessage(text) { + if (rejectPrompt) throw new Error("fixture: prompt refused"); + this.prompts.push(text); + this.idle = false; + }, + abort() { + this.aborts += 1; + this.idle = true; + for (const resolve of this.idleWaiters.splice(0)) resolve(); + return Promise.resolve(); + }, + get isIdle() { return this.idle; }, + }; + for (const name of missing) delete session[name]; + return session; +} + +function makeHost(session) { + const host = Object.create(InteractiveMode.prototype); + const children = []; + Object.assign(host, { + // Pi reads its session through runtimeHost, which a session replacement swaps. + runtimeHost: { session }, + compactionQueuedMessages: [], + statuses: [], + warnings: [], + editorText: "", + pendingMessagesContainer: { + clear() { children.length = 0; }, + addChild(child) { children.push(child); }, + }, + editor: { + getText: () => host.editorText, + setText: (text) => { host.editorText = text; }, + }, + getAppKeyDisplay: () => "Alt+Up", + showStatus(message) { host.statuses.push(message); }, + showWarning(message) { host.warnings.push(message); }, + rows() { + return children.flatMap((child) => child.render(200)).map(stripAnsi).join("\n"); + }, + }); + return host; +} + +const assertNoOperationalText = (text, context) => { + for (const needle of ["⁣", "FIRSTMATE_OP: v1 watcher", "QUEUED_MONITOR", "QUEUED_LEGACY_AWAY"]) { + check(!text.includes(needle), `${context} exposed operational text ${JSON.stringify(needle)}: ${JSON.stringify(text)}`); + } +}; + +visibility.setCalmPresentation(true); + +// 1. Supported session: hidden while queued, kept on Escape, delivered once in a new turn. +{ + const session = makeSession(); + const host = makeHost(session); + session.steering.push(watcherTwo); + session.followUp.push(watcherOne, captainText, legacyAway, lookalike); + host.updatePendingMessagesDisplay(); + const rows = host.rows(); + assertNoOperationalText(rows, "queued listing under Calm"); + check(rows.includes(`Follow-up: ${captainText}`), `captain's queued row disappeared: ${rows}`); + check(rows.includes(lookalike), `an unauthenticated lookalike was hidden: ${rows}`); + check(rows.includes("to edit all queued messages"), `dequeue hint missing: ${rows}`); + check(host.warnings.length === 0, `a supported session warned: ${host.warnings}`); + + host.editorText = "CAPTAIN_DRAFT"; + const restored = host.restoreQueuedMessagesToEditor({ abort: true }); + assertNoOperationalText(host.editorText, "editor after Escape"); + check(host.editorText === `${captainText}\n\n${lookalike}\n\nCAPTAIN_DRAFT`, `editor text changed: ${JSON.stringify(host.editorText)}`); + check(restored === 2, `restore reported ${restored} messages instead of the two captain-authored ones`); + check(session.aborts === 1, "Escape did not abort the run"); + assertNoOperationalText(host.rows(), "queued listing after Escape"); + + await settle(); + check(JSON.stringify(session.prompts) === JSON.stringify([watcherTwo]), `continuation prompt was ${JSON.stringify(session.prompts)}`); + check(JSON.stringify(session.steering) === "[]", `steering left behind: ${JSON.stringify(session.steering)}`); + check(JSON.stringify(session.followUp) === JSON.stringify([watcherOne, legacyAway]), `follow-ups lost their order: ${JSON.stringify(session.followUp)}`); + const delivered = [...session.prompts, ...session.steering, ...session.followUp]; + for (const text of operationalTexts) { + check(delivered.filter((value) => value === text).length === 1, `notification not kept exactly once: ${JSON.stringify(text)}`); + } + check(JSON.stringify(host.statuses) === JSON.stringify([layout.CALM_SUPERVISION_CONTINUES_NOTICE]), `continuation notice: ${JSON.stringify(host.statuses)}`); + assertNoOperationalText(host.statuses.join("\n"), "continuation notice"); +} + +// 2. The dequeue key restores captain text while the run keeps going: nothing restarts. +{ + const session = makeSession(); + const host = makeHost(session); + session.followUp.push(captainText, watcherOne); + host.updatePendingMessagesDisplay(); + const restored = host.restoreQueuedMessagesToEditor(); + check(restored === 1 && host.editorText === captainText, `dequeue restored ${restored}: ${JSON.stringify(host.editorText)}`); + check(JSON.stringify(session.followUp) === JSON.stringify([watcherOne]), `dequeue lost the notification: ${JSON.stringify(session.followUp)}`); + await settle(); + check(session.prompts.length === 0 && host.statuses.length === 0, "a dequeue while the run is still active started or announced a turn"); +} + +// 2b. Navigating the session tree during a run restores without abort, aborts the run, and +// then holds the session busy while it navigates: the hidden notification waits for the +// navigation to finish and is then delivered exactly once in a new turn. +{ + const session = makeSession(); + const host = makeHost(session); + session.followUp.push(captainText, watcherOne); + host.updatePendingMessagesDisplay(); + host.restoreQueuedMessagesToEditor(); + await session.abort(); + session.idle = false; + assertNoOperationalText(host.editorText, "editor after tree navigation"); + check(host.editorText === captainText, `tree navigation restored ${JSON.stringify(host.editorText)}`); + await settle(); + check(session.prompts.length === 0 && host.statuses.length === 0, `a turn started during tree navigation: ${JSON.stringify(session.prompts)}`); + session.idle = true; + for (const resolve of session.idleWaiters.splice(0)) resolve(); + await settle(); + check(JSON.stringify(session.prompts) === JSON.stringify([watcherOne]), `tree navigation continuation prompt was ${JSON.stringify(session.prompts)}`); + check(JSON.stringify(session.followUp) === "[]", `tree navigation left the notification queued: ${JSON.stringify(session.followUp)}`); + check(JSON.stringify(host.statuses) === JSON.stringify([layout.CALM_SUPERVISION_CONTINUES_NOTICE]), `tree navigation notice: ${JSON.stringify(host.statuses)}`); +} + +// 3. A row already hidden stays hidden on Escape even if the classifier cannot answer again. +{ + const session = makeSession(); + const host = makeHost(session); + session.followUp.push(watcherOne, captainText); + host.updatePendingMessagesDisplay(); + writeFileSync(process.env.FM_CLASSIFIER_DOWN, ""); + try { + host.restoreQueuedMessagesToEditor({ abort: true }); + } finally { + rmSync(process.env.FM_CLASSIFIER_DOWN, { force: true }); + } + assertNoOperationalText(host.editorText, "editor after Escape with the classifier down"); + await settle(); + check(JSON.stringify(session.prompts) === JSON.stringify([watcherOne]), `hidden notification not delivered: ${JSON.stringify(session.prompts)}`); +} + +// 4. Compaction-held notifications are kept but never reach the agent queue, so no turn +// starts and none is announced. +{ + const session = makeSession(); + const host = makeHost(session); + host.compactionQueuedMessages.push({ text: watcherOne, mode: "followUp" }, { text: captainText, mode: "followUp" }); + host.updatePendingMessagesDisplay(); + assertNoOperationalText(host.rows(), "compaction-queued listing"); + host.restoreQueuedMessagesToEditor({ abort: true }); + assertNoOperationalText(host.editorText, "editor after Escape during compaction"); + check(host.editorText === captainText, `captain compaction text not restored: ${JSON.stringify(host.editorText)}`); + check(JSON.stringify(host.compactionQueuedMessages) === JSON.stringify([{ text: watcherOne, mode: "followUp" }]), `compaction notification not kept: ${JSON.stringify(host.compactionQueuedMessages)}`); + await settle(); + check(session.prompts.length === 0, "compaction-only retention started a turn"); + check(host.statuses.length === 0, `compaction-only retention announced a turn: ${JSON.stringify(host.statuses)}`); +} + +// 5. A continuation Pi refuses to start puts the notification back instead of losing it. +{ + const session = makeSession({ rejectPrompt: true }); + const host = makeHost(session); + session.followUp.push(watcherOne); + host.updatePendingMessagesDisplay(); + host.restoreQueuedMessagesToEditor({ abort: true }); + await settle(); + check(JSON.stringify(session.followUp) === JSON.stringify([watcherOne]), `refused continuation dropped the notification: ${JSON.stringify(session.followUp)}`); +} + +// 6. Sessions missing any retention member: nothing is hidden, one warning, stock Escape. +for (const missing of ["_queueSteer", "_queueFollowUp", "sendUserMessage", "waitForIdle"]) { + const session = makeSession({ missing: [missing] }); + const host = makeHost(session); + session.followUp.push(watcherOne, captainText); + host.updatePendingMessagesDisplay(); + host.updatePendingMessagesDisplay(); + const rows = host.rows(); + check(rows.includes("QUEUED_MONITOR_ONE"), `session without ${missing} hid a row it cannot keep: ${rows}`); + check(JSON.stringify(host.warnings) === JSON.stringify([layout.CALM_QUEUED_ROWS_UNSUPPORTED_WARNING]), `session without ${missing} warned ${JSON.stringify(host.warnings)}`); + assertNoOperationalText(layout.CALM_QUEUED_ROWS_UNSUPPORTED_WARNING, "compatibility warning"); + host.restoreQueuedMessagesToEditor({ abort: true }); + check(host.editorText === `${watcherOne}\n\n${captainText}`, `session without ${missing} changed stock Escape: ${JSON.stringify(host.editorText)}`); + check(host.warnings.length === 1, `session without ${missing} warned again on Escape`); + await settle(); + check(session.prompts.length === 0 && host.statuses.length === 0, `session without ${missing} started a turn`); +} + +// 7. Calm off is stock; turning it off while rows are hidden keeps them out of the editor +// until the listing is redrawn, and the redraw follows the new choice. +{ + visibility.setCalmPresentation(false); + const session = makeSession(); + const host = makeHost(session); + session.followUp.push(watcherOne, captainText); + host.updatePendingMessagesDisplay(); + check(host.rows().includes("QUEUED_MONITOR_ONE"), "Calm off hid a queued row"); + host.restoreQueuedMessagesToEditor(); + check(host.editorText === `${watcherOne}\n\n${captainText}`, `Calm off changed stock dequeue: ${JSON.stringify(host.editorText)}`); + + visibility.setCalmPresentation(true); + const toggled = makeSession(); + const toggledHost = makeHost(toggled); + toggled.followUp.push(watcherOne, captainText); + toggledHost.updatePendingMessagesDisplay(); + visibility.setCalmPresentation(false); + toggledHost.restoreQueuedMessagesToEditor(); + assertNoOperationalText(toggledHost.editorText, "editor after Calm turned off over a hidden row"); + check(toggledHost.editorText === captainText, `captain text not restored after the toggle: ${JSON.stringify(toggledHost.editorText)}`); + check(JSON.stringify(toggled.followUp) === JSON.stringify([watcherOne]), `toggle-time Escape lost the notification: ${JSON.stringify(toggled.followUp)}`); + check(toggledHost.rows().includes("QUEUED_MONITOR_ONE"), "Calm off kept a queued row hidden after Pi redrew the listing"); + visibility.setCalmPresentation(true); + layout.refreshCalmPendingOperationalRows(); + assertNoOperationalText(toggledHost.rows(), "queued listing after turning Calm on"); + visibility.setCalmPresentation(false); + layout.refreshCalmPendingOperationalRows(); + check(toggledHost.rows().includes("QUEUED_MONITOR_ONE"), "turning Calm off did not redraw the hidden row"); +} +JS +) + status=$? + [ "$status" -eq 0 ] || fail "Pi Calm queued operational rows: $out" + [ -z "$out" ] || fail "Pi Calm queued-row test printed output: $out" + pass "Calm hides queued Firstmate rows only on a session that can keep them, keeps hidden ones out of the editor on Escape, delivers them once in order, and leaves unsupported sessions and Calm off stock" +} + test_builtin_gate_load_time() { local fixture out output_file status if ! command -v node >/dev/null 2>&1 || ! command -v npm >/dev/null 2>&1; then @@ -477,6 +796,7 @@ test_builtin_gate_load_time() { cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" cp "$PRESERVATION" "$fixture/project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$PENDING_OPERATIONAL_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship.ts" cp "$WORKING_SHIP_SPRITE" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" @@ -565,6 +885,7 @@ test_calm_activation_collision_and_regression_bound() { cp "$ASSISTANT_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-assistant-layout.ts" cp "$PRESERVATION" "$fixture/project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$PENDING_OPERATIONAL_LAYOUT" "$fixture/project/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" cp "$VISIBILITY" "$fixture/project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship.ts" cp "$WORKING_SHIP_SPRITE" "$fixture/project/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" @@ -781,6 +1102,7 @@ test_rendering_and_session_lifecycle() { cp "$ASSISTANT_LAYOUT" "$fixture/lib/fm-calm-assistant-layout.ts" cp "$PRESERVATION" "$fixture/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/lib/fm-calm-operational-user-layout.ts" + cp "$PENDING_OPERATIONAL_LAYOUT" "$fixture/lib/fm-calm-pending-operational-layout.ts" cp "$VISIBILITY" "$fixture/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/lib/fm-calm-working-ship.ts" cp "$WORKING_SHIP_SPRITE" "$fixture/lib/fm-calm-working-ship-sprite.ts" @@ -1500,6 +1822,7 @@ test_calm_mid_turn_working_notes() { cp "$ASSISTANT_LAYOUT" "$fixture/lib/fm-calm-assistant-layout.ts" cp "$PRESERVATION" "$fixture/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/lib/fm-calm-operational-user-layout.ts" + cp "$PENDING_OPERATIONAL_LAYOUT" "$fixture/lib/fm-calm-pending-operational-layout.ts" cp "$VISIBILITY" "$fixture/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/lib/fm-calm-working-ship.ts" cp "$WORKING_SHIP_SPRITE" "$fixture/lib/fm-calm-working-ship-sprite.ts" @@ -1806,6 +2129,7 @@ test_operational_followup_turn_e2e() { cp "$ASSISTANT_LAYOUT" "$project/.pi/extensions/lib/fm-calm-assistant-layout.ts" cp "$PRESERVATION" "$project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$PENDING_OPERATIONAL_LAYOUT" "$project/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" cp "$VISIBILITY" "$project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$project/.pi/extensions/lib/fm-calm-working-ship.ts" cp "$WORKING_SHIP_SPRITE" "$project/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" @@ -2153,6 +2477,202 @@ JS pass "Pi operational follow-up E2E processes exact user-role notifications once while Calm hides current and adjacent rows, Calm off and absent render them, and restart preserves semantics" } +# The real-Pi counterpart of test_queued_operational_rows: a watcher notification queued +# while a tool holds the turn, then Escape, exactly as a captain would press it. +test_queued_operational_escape_e2e() { + local project home config sessions version pane session_file i + if ! command -v pi >/dev/null 2>&1 || ! command -v tmux >/dev/null 2>&1; then + echo "skip: pi or tmux not found for Pi Calm queued-row Escape E2E" + return 0 + fi + version=$(pi --version 2>/dev/null || true) + record_pi_version_evidence "$version" "Pi Calm queued-row Escape E2E" + + project="$TMP_ROOT/queued-escape-project" + home="$TMP_ROOT/queued-escape-home" + config="$TMP_ROOT/queued-escape-config" + sessions="$TMP_ROOT/queued-escape-sessions" + mkdir -p "$project/.pi/extensions/lib" "$home/config" "$config" "$sessions" + fm_git_init_commit "$project" + cp "$EXT" "$project/.pi/extensions/fm-calm.ts" + cp "$ASSISTANT_LAYOUT" "$project/.pi/extensions/lib/fm-calm-assistant-layout.ts" + cp "$PRESERVATION" "$project/.pi/extensions/lib/fm-calm-preservation.ts" + cp "$OPERATIONAL_USER_LAYOUT" "$project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$PENDING_OPERATIONAL_LAYOUT" "$project/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" + cp "$VISIBILITY" "$project/.pi/extensions/lib/fm-calm-visibility.ts" + cp "$WORKING_SHIP" "$project/.pi/extensions/lib/fm-calm-working-ship.ts" + cp "$WORKING_SHIP_SPRITE" "$project/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" + cp "$PI_OPERATIONAL_INPUT" "$project/.pi/extensions/lib/fm-operational-input.ts" + printf '%s\n' '{"followUpMode":"all"}' >"$config/settings.json" + + cat >"$project/queued-escape-e2e.ts" <<'TS' +import { writeFileSync } from "node:fs"; +import { createFauxCore, fauxAssistantMessage, fauxText, fauxToolCall } from "@earendil-works/pi-ai"; +import type { ExtensionAPI } from "@earendil-works/pi-coding-agent"; +import { Type } from "typebox"; +import { encodeFirstmateOperationalInput } from "./.pi/extensions/lib/fm-operational-input.ts"; + +let label = ""; + +function lastUserText(messages: readonly { role: string; content: unknown }[]): string { + const user = [...messages].reverse().find((message) => message.role === "user"); + if (!user) return ""; + if (typeof user.content === "string") return user.content; + return (user.content as { type: string; text?: string }[]) + .filter((block) => block.type === "text") + .map((block) => block.text ?? "") + .join("\n"); +} + +export default function (pi: ExtensionAPI): void { + const faux = createFauxCore({ + api: "queued-escape-e2e-api", + provider: "queued-escape-e2e", + models: [{ + id: "deterministic", + name: "Calm queued-row Escape E2E", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 4096, + maxTokens: 128, + }], + tokenSize: { min: 1, max: 1 }, + }); + // The captain prompt holds the turn in a tool; a monitoring notification gets its own reply. + const respond = (context: { messages: readonly { role: string; content: unknown }[] }) => { + const text = lastUserText(context.messages); + if (text.includes(`MONITOR_${label}`)) return fauxAssistantMessage([fauxText(`MONITOR_HANDLED_${label}`)]); + if (context.messages[context.messages.length - 1]?.role === "user") { + return fauxAssistantMessage([fauxToolCall("hold_turn", {}, { id: `hold_${label}` })], { stopReason: "toolUse" }); + } + return fauxAssistantMessage([fauxText(`CAPTAIN_ANSWER_${label}`)]); + }; + pi.registerProvider("queued-escape-e2e", { + baseUrl: "http://127.0.0.1/unused", + apiKey: "test-only", + api: faux.api, + models: faux.models, + streamSimple: faux.streamSimple, + }); + pi.registerTool({ + name: "hold_turn", + label: "hold_turn", + description: "Hold the turn open until it is aborted.", + parameters: Type.Object({}), + async execute(_id, _params, signal) { + await pi.sendUserMessage( + encodeFirstmateOperationalInput("watcher", `MONITOR_${label}_ONE`), + { deliverAs: "followUp" }, + ); + writeFileSync(process.env.QUEUED_ESCAPE_HELD as string, label); + await new Promise((resolve) => signal?.addEventListener("abort", () => resolve(), { once: true })); + return { content: [{ type: "text", text: "released" }], details: {} }; + }, + }); + pi.registerCommand("queued-escape-e2e", { + description: "Hold one captain turn open while a monitoring notification queues.", + handler: async (args, ctx) => { + label = args.trim(); + const model = ctx.modelRegistry.find("queued-escape-e2e", "deterministic"); + if (!model || !(await pi.setModel(model))) throw new Error("queued-escape E2E model unavailable"); + faux.setResponses(Array.from({ length: 8 }, () => respond)); + pi.sendUserMessage(`CAPTAIN_PROMPT_${label}`); + }, + }); +} +TS + + run_queued_escape_case() { + local calm_state=$1 label=$2 captain_queued=$3 held="$TMP_ROOT/queued-escape-held-$2" + tmux -L "$TMUX_SOCKET" kill-session -t "$TMUX_SESSION" 2>/dev/null || true + printf '%s\n' "$calm_state" >"$home/config/calm" + mkdir -p "$sessions/$label" + tmux -L "$TMUX_SOCKET" new-session -d -s "$TMUX_SESSION" -x 160 -y 36 \ + "cd '$project' && env FM_HOME='$home' PI_CODING_AGENT_DIR='$config' FM_OPERATIONAL_INPUT_SCRIPT='$OPERATIONAL_INPUT' QUEUED_ESCAPE_HELD='$held' PI_OFFLINE=1 pi --approve --no-context-files --no-skills --no-prompt-templates --no-extensions -e ./.pi/extensions/fm-calm.ts -e ./queued-escape-e2e.ts --session-dir '$sessions/$label'; rc=\$?; printf '\nPI_EXIT=%s\n' \"\$rc\"; sleep 20" + wait_for_text "$TMP_ROOT/queued-escape-pane" 'queued-escape-e2e.ts' \ + || fail "Pi queued-row $label case did not reach the ready composer" + tmux -L "$TMUX_SOCKET" send-keys -t "$TMUX_SESSION" -l "/queued-escape-e2e $label" + tmux -L "$TMUX_SOCKET" send-keys -t "$TMUX_SESSION" Enter + i=0 + while [ ! -e "$held" ] && [ "$i" -lt 200 ]; do + sleep 0.05 + i=$((i + 1)) + done + [ -e "$held" ] || fail "Pi queued-row $label case never queued the monitoring notification" + if [ "$captain_queued" = yes ]; then + tmux -L "$TMUX_SOCKET" send-keys -t "$TMUX_SESSION" -l "CAPTAIN_QUEUED_$label" + tmux -L "$TMUX_SOCKET" send-keys -t "$TMUX_SESSION" M-Enter + wait_for_text "$TMP_ROOT/queued-escape-pane" "Follow-up: CAPTAIN_QUEUED_$label" \ + || fail "Pi queued-row $label case did not list the captain's queued follow-up" + elif [ "$calm_state" = on ]; then + # Nothing appears to wait for, so give Pi's listing a moment to repaint. + sleep 1 + tmux -L "$TMUX_SOCKET" capture-pane -p -t "$TMUX_SESSION" -S -600 >"$TMP_ROOT/queued-escape-pane" + else + wait_for_text "$TMP_ROOT/queued-escape-pane" "Follow-up:" \ + || fail "Pi queued-row $label case never listed the queued notification" + fi + pane=$(cat "$TMP_ROOT/queued-escape-pane") + if [ "$calm_state" = on ]; then + assert_not_contains "$pane" "MONITOR_${label}_ONE" "Pi Calm listed a queued Firstmate notification" + else + assert_contains "$pane" "Follow-up: ⁣FIRSTMATE_OP: v1 watcher: MONITOR_${label}_ONE" "Pi Calm off changed the stock queued listing" + fi + + tmux -L "$TMUX_SOCKET" send-keys -t "$TMUX_SESSION" Escape + if [ "$calm_state" = on ]; then + i=0 + while [ "$i" -lt 200 ]; do + session_file=$(find "$sessions/$label" -type f -name '*.jsonl' | head -1) + [ -n "$session_file" ] && grep -Fq "MONITOR_HANDLED_$label" "$session_file" && break + sleep 0.05 + i=$((i + 1)) + done + wait_for_text "$TMP_ROOT/queued-escape-pane" "MONITOR_HANDLED_$label" \ + || fail "Pi Calm did not deliver the notification kept across Escape" + pane=$(cat "$TMP_ROOT/queued-escape-pane") + assert_not_contains "$pane" "MONITOR_${label}_ONE" "Pi Calm exposed a hidden notification after Escape" + assert_not_contains "$pane" "FIRSTMATE_OP" "Pi Calm exposed operational text after Escape" + assert_contains "$pane" "Firstmate supervision continues in a new turn." "Pi Calm restarted a turn silently after Escape" + if [ "$captain_queued" = yes ]; then + [ "$(tmux -L "$TMUX_SOCKET" capture-pane -p -t "$TMUX_SESSION" | grep -c "^CAPTAIN_QUEUED_$label *\$")" -eq 1 ] \ + || fail "Pi Calm did not return the captain's queued text to the editor on Escape" + fi + else + wait_for_text "$TMP_ROOT/queued-escape-pane" "aborted" \ + || fail "Pi Calm off Escape did not abort the turn" + sleep 1 + tmux -L "$TMUX_SOCKET" capture-pane -p -t "$TMUX_SESSION" >"$TMP_ROOT/queued-escape-pane" + pane=$(cat "$TMP_ROOT/queued-escape-pane") + assert_contains "$pane" "FIRSTMATE_OP: v1 watcher: MONITOR_${label}_ONE" "Pi Calm off changed stock Escape, which restores every queued message to the editor" + session_file=$(find "$sessions/$label" -type f -name '*.jsonl' | head -1) + fi + + node - "$session_file" "$label" "$calm_state" <<'JS' || fail "Pi queued-row $label case persisted the wrong delivery" +const fs = require("node:fs"); +const [file, label, calm] = process.argv.slice(2); +const entries = fs.readFileSync(file, "utf8").trim().split("\n").map(JSON.parse); +const text = (content) => typeof content === "string" + ? content + : (content ?? []).filter((item) => item.type === "text").map((item) => item.text).join("\n"); +const users = entries.filter((entry) => entry.type === "message" && entry.message.role === "user").map((entry) => text(entry.message.content)); +const handled = entries.filter((entry) => entry.type === "message" && entry.message.role === "assistant" && text(entry.message.content) === `MONITOR_HANDLED_${label}`); +const notification = `⁣FIRSTMATE_OP: v1 watcher: MONITOR_${label}_ONE`; +const expected = calm === "on" ? 1 : 0; +if (users.filter((value) => value === notification).length !== expected) throw new Error(`notification delivered ${users.filter((value) => value === notification).length} times: ${JSON.stringify(users)}`); +if (handled.length !== expected) throw new Error(`notification handled ${handled.length} times`); +if (users.some((value) => value.includes(`CAPTAIN_QUEUED_${label}`))) throw new Error("captain's restored text was sent instead of returned to the editor"); +JS + tmux -L "$TMUX_SOCKET" kill-session -t "$TMUX_SESSION" 2>/dev/null || true + } + + run_queued_escape_case on queued_on no + run_queued_escape_case on queued_mixed yes + run_queued_escape_case off queued_off no + pass "Pi $version with Calm on keeps a queued Firstmate notification unlisted, out of the editor on Escape, and delivers it once in a new announced turn, while Calm off stays stock" +} + test_hidden_block_geometry_e2e() { local project home config sessions session_file snapshot expanded_snapshot calm_off_snapshot restarted_snapshot local version skill_line final_line gap i @@ -2182,6 +2702,7 @@ test_hidden_block_geometry_e2e() { cp "$ASSISTANT_LAYOUT" "$project/.pi/extensions/lib/fm-calm-assistant-layout.ts" cp "$PRESERVATION" "$project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$PENDING_OPERATIONAL_LAYOUT" "$project/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" cp "$VISIBILITY" "$project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$project/.pi/extensions/lib/fm-calm-working-ship.ts" cp "$WORKING_SHIP_SPRITE" "$project/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" @@ -2418,6 +2939,7 @@ test_working_ship_geometry_and_lifecycle() { cp "$ASSISTANT_LAYOUT" "$fixture/lib/fm-calm-assistant-layout.ts" cp "$PRESERVATION" "$fixture/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$fixture/lib/fm-calm-operational-user-layout.ts" + cp "$PENDING_OPERATIONAL_LAYOUT" "$fixture/lib/fm-calm-pending-operational-layout.ts" cp "$VISIBILITY" "$fixture/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$fixture/lib/fm-calm-working-ship.ts" cp "$WORKING_SHIP_SPRITE" "$fixture/lib/fm-calm-working-ship-sprite.ts" @@ -3449,6 +3971,7 @@ test_interactive_terminal_e2e() { cp "$ASSISTANT_LAYOUT" "$project/.pi/extensions/lib/fm-calm-assistant-layout.ts" cp "$PRESERVATION" "$project/.pi/extensions/lib/fm-calm-preservation.ts" cp "$OPERATIONAL_USER_LAYOUT" "$project/.pi/extensions/lib/fm-calm-operational-user-layout.ts" + cp "$PENDING_OPERATIONAL_LAYOUT" "$project/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" cp "$VISIBILITY" "$project/.pi/extensions/lib/fm-calm-visibility.ts" cp "$WORKING_SHIP" "$project/.pi/extensions/lib/fm-calm-working-ship.ts" cp "$WORKING_SHIP_SPRITE" "$project/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" @@ -4299,11 +4822,13 @@ test_home_resolution test_pi_compat_no_upper_bound test_pi_compat_degraded_adapter test_pi_compat_missing_adapter_exports +test_queued_operational_rows test_builtin_gate_load_time test_calm_activation_collision_and_regression_bound test_rendering_and_session_lifecycle test_calm_mid_turn_working_notes test_operational_followup_turn_e2e +test_queued_operational_escape_e2e test_hidden_block_geometry_e2e test_working_ship_geometry_and_lifecycle test_export_dom_render_guard diff --git a/tests/fm-calm-pi-queue-retention-live-e2e.test.sh b/tests/fm-calm-pi-queue-retention-live-e2e.test.sh new file mode 100755 index 00000000000..f2384e26398 --- /dev/null +++ b/tests/fm-calm-pi-queue-retention-live-e2e.test.sh @@ -0,0 +1,153 @@ +#!/usr/bin/env bash +# Default-on live guard for the capability Calm's queued-row adapter preflights: a real +# interactive Pi session must expose every session member the adapter uses to keep a hidden +# queued notification across Escape. Those members live on Pi's session object, not on an +# exported class, so only a running Pi can answer. +# +# When a member is missing, the adapter degrades quietly by design: queued Firstmate rows +# stay visible and one warning appears. This guard fails loudly naming the installed Pi +# version instead, so a Pi release that removes the capability is noticed rather than +# silently costing the captain the hidden rows. The Escape flow itself is pinned by +# test_queued_operational_rows and test_queued_operational_escape_e2e in +# tests/fm-calm-pi-extension.test.sh. +# +# No model turn reaches any provider: a local faux provider holds one turn in a tool so a +# message can queue, and a probe extension records the live session's members from Pi's +# own queued-listing redraw. Scratch FM_HOME, project, Pi agent directory, session +# directory, and a private tmux socket; nothing global is touched. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +fm_live_gate default-on FM_CALM_PI_QUEUE_RETENTION_LIVE pi tmux node + +PI_VERSION=$(pi --version 2>/dev/null || printf 'unknown') +SOCKET="fm-calm-queue-retention-$$" +SESSION=calm-queue-retention +TMP_ROOT=$(fm_test_tmproot fm-calm-queue-retention) +PROJECT="$TMP_ROOT/project" +PROBE_OUT="$TMP_ROOT/session-members.json" +mkdir -p "$PROJECT/.pi/extensions/lib" "$TMP_ROOT/home/config" "$TMP_ROOT/agent" "$TMP_ROOT/sessions" + +cleanup() { + tmux -L "$SOCKET" kill-server 2>/dev/null || true + fm_test_cleanup +} +trap cleanup EXIT + +fm_git_init_commit "$PROJECT" +cp "$ROOT/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" "$PROJECT/.pi/extensions/lib/" +cp "$ROOT/.pi/extensions/lib/fm-calm-visibility.ts" "$PROJECT/.pi/extensions/lib/" +cp "$ROOT/.pi/extensions/lib/fm-operational-input.ts" "$PROJECT/.pi/extensions/lib/" + +cat >"$PROJECT/queue-retention-probe.ts" <<'TS' +import { writeFileSync } from "node:fs"; +import { createFauxCore, fauxAssistantMessage, fauxText, fauxToolCall } from "@earendil-works/pi-ai"; +import * as PiCodingAgent from "@earendil-works/pi-coding-agent"; +import type { ExtensionAPI } from "@earendil-works/pi-coding-agent"; +import { Type } from "typebox"; +import { CALM_QUEUE_RETENTION_SESSION_METHODS } from "./.pi/extensions/lib/fm-calm-pending-operational-layout.ts"; + +const out = process.env.QUEUE_RETENTION_PROBE_OUT as string; + +export default function (pi: ExtensionAPI): void { + const prototype = (PiCodingAgent.InteractiveMode as unknown as { prototype: Record }).prototype; + const prototypeMembers = Object.fromEntries( + ["getAllQueuedMessages", "updatePendingMessagesDisplay", "clearAllQueues", "restoreQueuedMessagesToEditor"] + .map((name) => [name, typeof prototype[name]]), + ); + const original = prototype.updatePendingMessagesDisplay as (this: Record) => void; + prototype.updatePendingMessagesDisplay = function (this: Record): void { + const session = this.session as Record | undefined; + if (session && (session.getFollowUpMessages as () => string[])().length > 0) { + writeFileSync(out, JSON.stringify({ + prototype: prototypeMembers, + session: Object.fromEntries(CALM_QUEUE_RETENTION_SESSION_METHODS.map((name) => [name, typeof session[name]])), + isIdle: typeof session.isIdle, + compactionQueuedMessages: Array.isArray(this.compactionQueuedMessages), + })); + } + original.call(this); + }; + + const faux = createFauxCore({ + api: "queue-retention-probe-api", + provider: "queue-retention-probe", + models: [{ + id: "deterministic", + name: "Calm queue-retention capability probe", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 4096, + maxTokens: 128, + }], + tokenSize: { min: 1, max: 1 }, + }); + pi.registerProvider("queue-retention-probe", { + baseUrl: "http://127.0.0.1/unused", + apiKey: "test-only", + api: faux.api, + models: faux.models, + streamSimple: faux.streamSimple, + }); + pi.registerTool({ + name: "hold_turn", + label: "hold_turn", + description: "Queue one follow-up, then hold the turn until it is aborted.", + parameters: Type.Object({}), + async execute(_id, _params, signal) { + await pi.sendUserMessage("QUEUE_RETENTION_PROBE_FOLLOW_UP", { deliverAs: "followUp" }); + await new Promise((resolve) => signal?.addEventListener("abort", () => resolve(), { once: true })); + return { content: [{ type: "text", text: "released" }], details: {} }; + }, + }); + pi.registerCommand("queue-retention-probe", { + description: "Hold a turn open while one follow-up queues.", + handler: async (_args, ctx) => { + const model = ctx.modelRegistry.find("queue-retention-probe", "deterministic"); + if (!model || !(await pi.setModel(model))) throw new Error("probe model unavailable"); + faux.setResponses([ + fauxAssistantMessage([fauxToolCall("hold_turn", {}, { id: "hold_probe" })], { stopReason: "toolUse" }), + fauxAssistantMessage([fauxText("QUEUE_RETENTION_PROBE_DONE")]), + ]); + pi.sendUserMessage("QUEUE_RETENTION_PROBE_PROMPT"); + }, + }); +} +TS + +tmux -L "$SOCKET" new-session -d -s "$SESSION" -x 160 -y 36 \ + "cd '$PROJECT' && env FM_HOME='$TMP_ROOT/home' PI_CODING_AGENT_DIR='$TMP_ROOT/agent' QUEUE_RETENTION_PROBE_OUT='$PROBE_OUT' PI_OFFLINE=1 pi --approve --no-context-files --no-skills --no-prompt-templates --no-extensions -e ./queue-retention-probe.ts --session-dir '$TMP_ROOT/sessions'; sleep 30" + +i=0 +until tmux -L "$SOCKET" capture-pane -p -t "$SESSION" 2>/dev/null | grep -Fq 'queue-retention-probe.ts'; do + i=$((i + 1)) + [ "$i" -lt 200 ] || fail "Pi $PI_VERSION did not reach its composer: $(tmux -L "$SOCKET" capture-pane -p -t "$SESSION" 2>/dev/null)" + sleep 0.05 +done +tmux -L "$SOCKET" send-keys -t "$SESSION" -l '/queue-retention-probe' +tmux -L "$SOCKET" send-keys -t "$SESSION" Enter +i=0 +until [ -s "$PROBE_OUT" ]; do + i=$((i + 1)) + [ "$i" -lt 200 ] || fail "Pi $PI_VERSION never redrew its queued listing for a queued follow-up: $(tmux -L "$SOCKET" capture-pane -p -t "$SESSION" 2>/dev/null)" + sleep 0.05 +done +tmux -L "$SOCKET" send-keys -t "$SESSION" Escape + +# shellcheck disable=SC2016 # Literal JavaScript; its template expressions are not shell expansions. +missing=$(node -e ' +const probe = JSON.parse(require("node:fs").readFileSync(process.argv[1], "utf8")); +const missing = []; +for (const [name, type] of Object.entries(probe.prototype)) if (type !== "function") missing.push(`InteractiveMode.${name}`); +for (const [name, type] of Object.entries(probe.session)) if (type !== "function") missing.push(`session.${name}`); +if (probe.isIdle !== "boolean") missing.push("session.isIdle"); +if (!probe.compactionQueuedMessages) missing.push("InteractiveMode.compactionQueuedMessages"); +if (Object.keys(probe.session).length === 0) missing.push("(no session members were probed)"); +process.stdout.write(missing.join(", ")); +' "$PROBE_OUT") || fail "could not read the Pi $PI_VERSION capability probe" +[ -z "$missing" ] \ + || fail "Pi $PI_VERSION lacks the queue-retention capability Calm needs to hide queued Firstmate rows: $missing" +pass "Pi $PI_VERSION exposes every queue-retention member Calm preflights before hiding queued Firstmate rows" diff --git a/tests/fm-pi-primary-live-e2e.test.sh b/tests/fm-pi-primary-live-e2e.test.sh index 3bf1d2a54ea..00adeb2b3a9 100755 --- a/tests/fm-pi-primary-live-e2e.test.sh +++ b/tests/fm-pi-primary-live-e2e.test.sh @@ -254,6 +254,7 @@ cp "$ROOT/.pi/extensions/fm-calm.ts" "$PROJECT/.pi/extensions/fm-calm.ts" cp "$ROOT/.pi/extensions/fm-primary-pi-watch.ts" "$PROJECT/.pi/extensions/fm-primary-pi-watch.ts" cp "$ROOT/.pi/extensions/lib/fm-calm-assistant-layout.ts" "$PROJECT/.pi/extensions/lib/fm-calm-assistant-layout.ts" cp "$ROOT/.pi/extensions/lib/fm-calm-operational-user-layout.ts" "$PROJECT/.pi/extensions/lib/fm-calm-operational-user-layout.ts" +cp "$ROOT/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" "$PROJECT/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" cp "$ROOT/.pi/extensions/lib/fm-calm-visibility.ts" "$PROJECT/.pi/extensions/lib/fm-calm-visibility.ts" cp "$ROOT/.pi/extensions/lib/fm-calm-working-ship.ts" "$PROJECT/.pi/extensions/lib/fm-calm-working-ship.ts" cp "$ROOT/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" "$PROJECT/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" diff --git a/tests/fm-pi-primary-types.test.sh b/tests/fm-pi-primary-types.test.sh index 4daef32b62b..2cb45a52f24 100755 --- a/tests/fm-pi-primary-types.test.sh +++ b/tests/fm-pi-primary-types.test.sh @@ -38,6 +38,7 @@ cp "$ROOT/.pi/extensions/lib/fm-branch-model-picker.ts" "$TMP_ROOT/lib/fm-branch cp "$ROOT/.pi/extensions/lib/fm-calm-assistant-layout.ts" "$TMP_ROOT/lib/fm-calm-assistant-layout.ts" cp "$ROOT/.pi/extensions/lib/fm-calm-preservation.ts" "$TMP_ROOT/lib/fm-calm-preservation.ts" cp "$ROOT/.pi/extensions/lib/fm-calm-operational-user-layout.ts" "$TMP_ROOT/lib/fm-calm-operational-user-layout.ts" +cp "$ROOT/.pi/extensions/lib/fm-calm-pending-operational-layout.ts" "$TMP_ROOT/lib/fm-calm-pending-operational-layout.ts" cp "$ROOT/.pi/extensions/lib/fm-calm-visibility.ts" "$TMP_ROOT/lib/fm-calm-visibility.ts" cp "$ROOT/.pi/extensions/lib/fm-calm-working-ship.ts" "$TMP_ROOT/lib/fm-calm-working-ship.ts" cp "$ROOT/.pi/extensions/lib/fm-calm-working-ship-sprite.ts" "$TMP_ROOT/lib/fm-calm-working-ship-sprite.ts" From 756f64ebc0f34e2ae1304fdda3e419b6cfd3bd71 Mon Sep 17 00:00:00 2001 From: Tiago Date: Fri, 25 Sep 2026 03:19:38 -0300 Subject: [PATCH 10/84] fix(bin): refuse teardown when a required source disappears (#5548) * fix(bin): refuse teardown when a required source disappears A missing sibling was sourced after cleanup had started, so Bash 3.2 exited 0 from the EXIT trap and Bash 5 continued and reported success. * no-mistakes(review): Remove unused FM_TEST_ONLY hook from teardown tests * no-mistakes(review): Check task backend sources before any teardown cleanup * test(gotmp): give teardown fixtures every tmux adapter sibling Teardown now refuses when a sibling the recorded backend's adapter sources is missing, so the fake bin must carry fm-session-lock-lib.sh, fm-agent-process-lib.sh and fm-gemini-lib.sh. --- bin/fm-backend.sh | 41 +++++++-- bin/fm-teardown.sh | 55 +++++++++++- tests/fm-gotmp.test.sh | 9 +- tests/fm-teardown.test.sh | 180 ++++++++++++++++++++++++++++++++++++++ 4 files changed, 274 insertions(+), 11 deletions(-) diff --git a/bin/fm-backend.sh b/bin/fm-backend.sh index ac7f73aa84b..f4fdde29436 100644 --- a/bin/fm-backend.sh +++ b/bin/fm-backend.sh @@ -613,15 +613,44 @@ fm_backend_expected_label_of_selector() { # # Each adapter is an independently linted canonical root. The /dev/null source # boundaries keep runtime dispatch from importing all five adapter ASTs into # every dispatcher consumer while preserving the runtime source operations. +# Bash 3.2 can enter an EXIT trap with status 0 after `set -e` aborts on a +# missing or unreadable dot-sourced file, and a newer Bash can print that +# diagnostic and keep going. Both report a successful teardown. Prove the +# adapter and the siblings it sources are readable regular files before `.`. +fm_backend_source_readable() { # + [ -f "$1" ] && [ -r "$1" ] +} + fm_backend_source() { # - local name=$1 adapter + local name=$1 adapter rel path siblings fm_backend_validate "$name" || return 1 adapter="$FM_BACKEND_LIB_DIR/backends/$name.sh" - # Bash 3.2 can enter an EXIT trap with status 0 after `set -e` aborts on a - # missing or unreadable dot-sourced file. Refuse the adapter explicitly so - # callers retain the real failure status and never continue a destructive - # lifecycle operation after an unavailable backend prerequisite. - [ -f "$adapter" ] && [ -r "$adapter" ] || return 1 + case "$name" in + tmux) + siblings="fm-tmux-lib.sh fm-composer-lib.sh fm-cursor-lib.sh fm-session-lock-lib.sh fm-agent-process-lib.sh fm-gemini-lib.sh" + ;; + herdr) + siblings="fm-composer-lib.sh fm-transition-lib.sh fm-agent-process-lib.sh fm-session-lock-lib.sh fm-gemini-lib.sh" + ;; + zellij) + siblings="fm-backend-hometag-lib.sh fm-composer-lib.sh" + ;; + orca) + siblings="fm-composer-lib.sh" + ;; + cmux) + siblings="fm-backend-hometag-lib.sh fm-composer-lib.sh" + ;; + *) + return 1 + ;; + esac + fm_backend_source_readable "$adapter" || return 1 + # shellcheck disable=SC2086 # sibling names are a fixed space-separated list + for rel in $siblings; do + path="$FM_BACKEND_LIB_DIR/$rel" + fm_backend_source_readable "$path" || return 1 + done case "$name" in tmux) if [ -z "${_FM_BACKEND_TMUX_SOURCED:-}" ]; then diff --git a/bin/fm-teardown.sh b/bin/fm-teardown.sh index 26f5bb84709..85aabb7c5d0 100755 --- a/bin/fm-teardown.sh +++ b/bin/fm-teardown.sh @@ -287,6 +287,51 @@ CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" SECONDMATE_REG="$DATA/secondmates.md" SUB_HOME_MARKER=".fm-secondmate-home" SUB_HOME_PARENT_MARKER=".fm-secondmate-parent" +# A missing `.` target is not a teardown result. Stock Bash 3.2 can abort it +# into an EXIT trap whose status is 0, and a newer Bash can print the +# diagnostic and continue into cleanup. Refuse by name before sourcing. +teardown_require_source() { # + if [ ! -f "$1" ] || [ ! -r "$1" ]; then + echo "error: teardown refused: required source $(basename "$1") is missing or unreadable; nothing was changed" >&2 + exit 1 + fi +} + +teardown_require_backend_prerequisites() { # + local backend=$1 task_id=$2 + if ! fm_backend_source "$backend"; then + echo "error: teardown refused: required $backend source is missing or unreadable for $task_id; nothing was changed" >&2 + return 1 + fi +} +for _teardown_source in \ + fm-tasks-axi-lib.sh \ + fm-backlog-transition-lib.sh \ + fm-timeout-lib.sh \ + fm-backend.sh \ + fm-control-lib.sh \ + fm-lock-lib.sh \ + fm-classify-lib.sh \ + fm-gate-refuse-lib.sh \ + fm-pr-lib.sh \ + fm-public-followup-lib.sh \ + fm-x-lib.sh \ + fm-env-lib.sh \ + fm-secondmate-registry-lib.sh \ + fm-secondmate-parent-lib.sh \ + fm-pending-reply-lib.sh \ + fm-operational-input.sh \ + fm-marker-lib.sh \ + fm-tmux-lib.sh \ + fm-composer-lib.sh \ + fm-cursor-lib.sh \ + fm-nm-run-lib.sh \ + fm-wake-lib.sh \ + fm-lease-lib.sh +do + teardown_require_source "$SCRIPT_DIR/$_teardown_source" +done +unset _teardown_source # shellcheck source=bin/fm-tasks-axi-lib.sh . "$SCRIPT_DIR/fm-tasks-axi-lib.sh" # shellcheck source=bin/fm-backlog-transition-lib.sh @@ -1053,6 +1098,10 @@ else T=$FM_BACKEND_VALIDATED_TARGET [ "$BACKEND" != orca ] || T_ORCA=$T fi +# The recorded backend, including every sibling its adapter sources, has to +# be readable before the first destructive step. --force does not override +# this. A forced descendant is proved in validate_firstmate_home_children_removal. +teardown_require_backend_prerequisites "$BACKEND" "$ID" || exit 1 if [ "${FM_TEARDOWN_GUARD_DONE:-0}" != 1 ]; then "$FM_ROOT/bin/fm-guard.sh" || true fi @@ -2927,6 +2976,7 @@ validate_firstmate_home_children_removal() { child_kind=$(meta_value "$child_meta" kind) [ -n "$child_kind" ] || child_kind=ship child_backend=$(fm_backend_of_meta "$child_meta") + teardown_require_backend_prerequisites "$child_backend" "$child_id" || return 1 if [ "$child_kind" = secondmate ]; then child_home=$(meta_value "$child_meta" home) [ -n "$child_home" ] || child_home=$child_wt @@ -2972,10 +3022,7 @@ FMEOF teardown_herdr_require_prerequisites() { # local task_id=$1 prerequisite - if ! fm_backend_source herdr; then - echo "error: herdr teardown prerequisites are unavailable for $task_id; nothing was changed - restore the adapter and rerun teardown" >&2 - return 1 - fi + teardown_require_backend_prerequisites herdr "$task_id" || return 1 for prerequisite in \ fm_backend_herdr_parse_target \ fm_backend_herdr_pane_presence_state \ diff --git a/tests/fm-gotmp.test.sh b/tests/fm-gotmp.test.sh index d3337fbc57c..41a8486fc88 100755 --- a/tests/fm-gotmp.test.sh +++ b/tests/fm-gotmp.test.sh @@ -51,12 +51,16 @@ make_fake_root() { # Symlink the REAL teardown so the test exercises actual code, not a copy. ln -s "$TEARDOWN" "$fake/bin/fm-teardown.sh" # fm-backend.sh is real, while its adapter is stubbed so this temp-cleanup - # test cannot depend on or mutate a host tmux server. + # test cannot depend on or mutate a host tmux server. Teardown still refuses + # unless every sibling the real tmux adapter sources is present. ln -s "$ROOT/bin/fm-backend.sh" "$fake/bin/fm-backend.sh" cat > "$fake/bin/backends/tmux.sh" <<'SH' fm_backend_tmux_kill() { return 0; } SH ln -s "$ROOT/bin/fm-tmux-lib.sh" "$fake/bin/fm-tmux-lib.sh" + ln -s "$ROOT/bin/fm-session-lock-lib.sh" "$fake/bin/fm-session-lock-lib.sh" + ln -s "$ROOT/bin/fm-agent-process-lib.sh" "$fake/bin/fm-agent-process-lib.sh" + ln -s "$ROOT/bin/fm-gemini-lib.sh" "$fake/bin/fm-gemini-lib.sh" ln -s "$ROOT/bin/fm-cursor-lib.sh" "$fake/bin/fm-cursor-lib.sh" ln -s "$ROOT/bin/fm-composer-lib.sh" "$fake/bin/fm-composer-lib.sh" ln -s "$ROOT/bin/fm-nm-run-lib.sh" "$fake/bin/fm-nm-run-lib.sh" @@ -161,6 +165,9 @@ test_teardown_skips_gracefully_without_tasktmp() { fm_backend_tmux_kill() { return 0; } SH ln -s "$ROOT/bin/fm-tmux-lib.sh" "$fake/bin/fm-tmux-lib.sh" + ln -s "$ROOT/bin/fm-session-lock-lib.sh" "$fake/bin/fm-session-lock-lib.sh" + ln -s "$ROOT/bin/fm-agent-process-lib.sh" "$fake/bin/fm-agent-process-lib.sh" + ln -s "$ROOT/bin/fm-gemini-lib.sh" "$fake/bin/fm-gemini-lib.sh" ln -s "$ROOT/bin/fm-cursor-lib.sh" "$fake/bin/fm-cursor-lib.sh" ln -s "$ROOT/bin/fm-composer-lib.sh" "$fake/bin/fm-composer-lib.sh" ln -s "$ROOT/bin/fm-nm-run-lib.sh" "$fake/bin/fm-nm-run-lib.sh" diff --git a/tests/fm-teardown.test.sh b/tests/fm-teardown.test.sh index 229bd7351fb..bbd1a65e7c6 100755 --- a/tests/fm-teardown.test.sh +++ b/tests/fm-teardown.test.sh @@ -3884,6 +3884,186 @@ EOF pass "the run abort and the leaked-process reap both complete before the destructive worktree return" } +# Copy the public teardown script tree, then drop or blank one required file. +# Symlinks keep the copy cheap; an unreadable case replaces one link with a +# real mode-000 file so the probe is of the file itself. +prepare_teardown_source_copy() { # + local case_dir=$1 f base dest="$1/test-root/bin" s + mkdir -p "$dest/backends" + for f in "$ROOT"/bin/*; do + base=$(basename "$f") + if [ -d "$f" ]; then + mkdir -p "$dest/$base" + for s in "$f"/*; do + ln -s "$s" "$dest/$base/$(basename "$s")" + done + else + ln -s "$f" "$dest/$base" + fi + done + printf 'manual\n' > "$case_dir/config/backlog-backend" + cat > "$case_dir/fakebin/treehouse" <> "$case_dir/treehouse.log" +exit 0 +SH + chmod +x "$case_dir/fakebin/treehouse" + : > "$case_dir/treehouse.log" + : > "$case_dir/state/task-x1.status" +} + +run_copied_teardown() { # [args...] + local case_dir=$1 + shift + FM_ROOT_OVERRIDE="$ROOT" \ + FM_STATE_OVERRIDE="$case_dir/state" \ + FM_DATA_OVERRIDE="$case_dir/data" \ + FM_CONFIG_OVERRIDE="$case_dir/config" \ + PATH="$case_dir/fakebin:$PATH" \ + "$case_dir/test-root/bin/fm-teardown.sh" task-x1 "$@" +} + +assert_source_refusal_preserved_state() { #