diff --git a/.agents/skills/afk/SKILL.md b/.agents/skills/afk/SKILL.md index 68b7bf1b7cd..f82de256fb8 100644 --- a/.agents/skills/afk/SKILL.md +++ b/.agents/skills/afk/SKILL.md @@ -163,13 +163,15 @@ The daemon still clears its buffer only on the backend's `empty` success verdict The daemon wraps `fm-watch.sh`, runs the watcher as a child, presents every durable wake after each actionable watcher close, classifies each presented record in bash, and acknowledges the presented generation only after routing completes. It self-handles the routine majority without consuming a firstmate turn. -Captain-relevant events, plus a bounded recheck of a declared external wait that is still declared, escalate to firstmate's context as one pre-read, single-line, batched digest. +Events selected by the routing below escalate to firstmate's context as one pre-read, single-line, batched digest. The digest is byte-bounded so every transport can carry it; when it cuts an event or omits events past its budget, it names a `state/.subsuper-digests/` file that holds every buffered event verbatim, so read that file before acting on a cut event. The captain-relevant verb set, declared-wait vocabulary, status-span classifier, and presentation-marker contract live in shared `bin/fm-classify-lib.sh`, while each supervisor owns its routing and fleet scan as a consumer of that policy. While `state/.afk` exists the daemon owns the watcher, so the watcher reverts to one-shot and lets the daemon do the triage - the two never run their triage at the same time. -Classify each wake this way: +Classify each wake this way, applying the steering-inbox exception before status-based routing: +- `stale` whose detail begins `unread firstmate instruction: stuck-busy ` or `steering-inbox busy bookkeeping unwritable: ` -> buffer the explicit inbox escalation for supervision in away and quiet mode, without consuming worker status or entering transient-stale recovery. + [`bin/fm-task-inbox-lib.sh`](../../../bin/fm-task-inbox-lib.sh) owns the busy budget and bookkeeping contract; `tests/fm-daemon.test.sh` covers this consumer boundary. - `signal` whose newly classified status span contains captain-relevant events -> escalate every event in source order. A nonterminal progress verb remains nonterminal even when its prose contains a legacy free-text token such as `PR ready`, `checks green`, `ready in branch`, or `merged`; only a bare legacy line with such a token escalates. Other signals with no captain-relevant event in the span -> self-handle. diff --git a/.pi/extensions/fm-branch-supervision.ts b/.pi/extensions/fm-branch-supervision.ts index 25f2e31d924..dc44e5d7c38 100644 --- a/.pi/extensions/fm-branch-supervision.ts +++ b/.pi/extensions/fm-branch-supervision.ts @@ -653,10 +653,16 @@ export default function (pi: ExtensionAPI) { // queued for the captain's next prompt. The durable truth is the store's // processed marker; this only paces re-presentation and resets with the // session generation. - type ProcessingState = { sequences: string; through: number; triggered: number; pending: boolean; nextTurnQueued: boolean }; + type ProcessingState = { sequences: string; through: number; triggered: number; pending: boolean; nextTurnQueued: boolean; visibleFinals: Set }; let processing: ProcessingState | null = null; let queuedProcessingContent: string | null = null; let processingOpenedThisRun = false; + // Compare only replies to the same consumed sequence set. A retry can be + // the first real handling, so only an empty or exact-repeat final is hidden. + // Buffer retry streaming until message_end can make that decision; tool + // messages always keep their prose, and a user message ends this scope. + let activeProcessing: { request: ProcessingState; retry: boolean } | null = null; + let userMessageThisTurn = false; let processedInitializedGeneration = -1; // One revision for BOTH selections: a model or effort change invalidates an // in-flight branch build exactly the same way. @@ -1095,7 +1101,7 @@ export default function (pi: ExtensionAPI) { } if (processing?.pending) return true; if (!processing || processing.sequences !== sequences) { - processing = { sequences, through, triggered: 0, pending: false, nextTurnQueued: false }; + processing = { sequences, through, triggered: 0, pending: false, nextTurnQueued: false, visibleFinals: new Set() }; } // A presentation already sent is consumed by the run it joins or opens; // until that run settles, sending a widened or identical copy would hand @@ -1677,7 +1683,6 @@ ${context.command} // duplicate suppression. Operational extension injections are not dialog. const prompt = event.prompt; processingOpenedThisRun = queuedProcessingContent !== null && prompt === queuedProcessingContent; - if (processingOpenedThisRun) queuedProcessingContent = null; const trimmed = prompt.trim(); if (!trimmed || isOperationalUserText(trimmed)) return; const file = currentMainSession.getSessionFile() ?? ""; @@ -1692,6 +1697,48 @@ ${context.command} // so a fresh copy may be queued again once this run settles unacknowledged. if (processing) processing.nextTurnQueued = false; }); + pi.on?.("turn_start", () => { + userMessageThisTurn = false; + }); + pi.on?.("message_start", (event) => { + if (event.message.role === "user") { + userMessageThisTurn = true; + activeProcessing = null; + } else if ( + event.message.role === "custom" && + isProcessingCustomMessage(event.message) && + queuedProcessingContent !== null && + event.message.content === queuedProcessingContent + ) { + // message_start covers both an idle custom prompt and a follow-up + // consumed inside an existing run; neither needs before_agent_start. + activeProcessing = !userMessageThisTurn && processing ? { request: processing, retry: processing.triggered > 1 } : null; + queuedProcessingContent = null; + } + }); + pi.registerMarkdownTransformer?.((markdown, context) => + activeProcessing?.retry && context.isStreaming && context.messageType !== "user" ? "" : markdown, + ); + pi.on?.("message_end", (event) => { + if (!activeProcessing || event.message.role !== "assistant") return; + // message_end runs before tool execution. Keep the whole message when + // it carries a call, including prose alongside fm_branch_processed. + if (event.message.content.some((part) => part.type === "toolCall")) return; + const text = event.message.content.filter((part) => part.type === "text").map((part) => part.text).join("\n").trim(); + const { request, retry } = activeProcessing; + if (!retry || (text && !request.visibleFinals.has(text))) { + if (text) request.visibleFinals.add(text); + return; + } + // Pi applies the replacement before persistence and transcript rendering. + // Preserve the message envelope, including provider usage accounting. + return { + message: { + ...event.message, + content: [], + }, + }; + }); pi.on?.("context", (event, ctx) => { if (!afkPostureRecordPresent(state)) return; const messages = event.messages ?? []; @@ -1714,6 +1761,7 @@ ${context.command} mainStreaming = false; queuedProcessingContent = null; processingOpenedThisRun = false; + activeProcessing = null; if (processing) processing.pending = false; const settledGeneration = generation; await enqueueDelivery(async () => { @@ -1773,6 +1821,8 @@ ${context.command} consecutiveProviderErrors = 0; providerRecovery = null; generation += 1; + activeProcessing = null; + userMessageThisTurn = false; mirrorCollection.collectAnchor = null; mirrorCollection.pendingCursor = null; mirrorCollection.stagedCaptain = null; @@ -1821,6 +1871,9 @@ ${context.command} shuttingDown = true; generation += 1; processing = null; + queuedProcessingContent = null; + activeProcessing = null; + userMessageThisTurn = false; pendingMirror.length = 0; currentMainSession = null; mirrorCollection.collectAnchor = null; @@ -2339,6 +2392,9 @@ ${context.command} }; } const remaining = await readUnprocessedOutcomes(acknowledgedGeneration); + if (acknowledgedGeneration === generation && activeProcessing && through >= activeProcessing.request.through) { + activeProcessing = null; + } if (remaining !== null && remaining.length === 0) processing = null; const open = remaining === null ? "the remaining outcomes could not be read" diff --git a/bin/fm-brief.sh b/bin/fm-brief.sh index 2ff9c727762..871673d56bd 100755 --- a/bin/fm-brief.sh +++ b/bin/fm-brief.sh @@ -661,7 +661,9 @@ If the top-level path is the primary checkout or not the worktree you were launc # Rules $RULE1 -2. Stay inside this worktree; modify nothing outside it. +2. Keep project edits inside this worktree; keep proof and scratch output outside it, under \`$DATA/$ID/\` or a temporary directory. + Outside the worktree, write only that task material and the status and steering-inbox records authorized below. + Leave the worktree clean before reporting done. 3. Use gh-axi for GitHub operations and chrome-devtools-axi for browser operations. 4. Report status by appending one line: \`$STATUS_APPEND\` diff --git a/bin/fm-procevent-quota.sh b/bin/fm-procevent-quota.sh index 16ce34da2c4..5c41aa1104e 100755 --- a/bin/fm-procevent-quota.sh +++ b/bin/fm-procevent-quota.sh @@ -17,7 +17,10 @@ # registered through `bin/fm-procevent.sh register`. # poll The blocking child the generic runner executes; never run this # directly in a conversational turn. It polls `quota-axi --json` -# until quota drops below the threshold or an error stops the watch. +# until quota drops below the threshold, invalid quota data stops +# the watch, or three consecutive transient command failures stop +# it. Missing or incompatible tools stop it immediately, and a +# successful read resets the command-failure streak. # classify Print the captured outcome class: low, exhausted, error, or unknown. # terminal Every quota poll is terminal because the source fires at most once. # source-id Print the canonical source id. @@ -52,6 +55,9 @@ STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" DEFAULT_INTERVAL=60 DEFAULT_THRESHOLD=10 +# Consecutive transient quota-axi read failures before poll goes terminal. +# Missing and incompatible tools bypass this budget. No config knob on purpose. +MAX_CONSECUTIVE_READ_FAILURES=3 SOURCE_ID_BASE=quota @@ -98,16 +104,43 @@ valid_percent() { } # quota_json [timeout] -# Run `quota-axi --json` bounded by the given timeout. A missing or incompatible -# quota-axi is an error condition, not a signal to fire. +# Run `quota-axi --json` bounded by the given timeout. +# Exit status: 0 prints JSON; 1 timed out; 2 missing; 3 incompatible; 4 other failure. +# A missing or incompatible quota-axi is an error condition, not a signal to fire. +# Callers tolerate a bounded streak of 1/4 before going terminal; 2/3 stay distinct. +# Each path probes --version once, validates that captured text through +# fm_quota_axi_version_compatible, then probes --json once. +# A slow or failing probe stays 1/4; "incompatible" is reserved for an actual +# unsupported or unparseable version string. quota_json() { - local timeout=${1:-} output + local timeout=${1:-} output rc=0 + if ! command -v quota-axi >/dev/null 2>&1; then + return 2 + fi if [ -n "$timeout" ]; then - fm_quota_axi_compatible "$timeout" >/dev/null 2>&1 || return 2 - output=$(fm_run_timed "$timeout" quota-axi --json 2>/dev/null /dev/null /dev/null /dev/null 2>&1 || return 2 - output=$(quota-axi --json 2>/dev/null /dev/null /dev/null +# True when the printed `quota-axi --version` text meets FM_QUOTA_AXI_MIN. +# Callers that already captured a bounded --version pass that text here so they +# do not launch a second, unbounded version probe. +fm_quota_axi_version_compatible() { + local output=${1-} parts major minor patch extra local min_major min_minor min_patch min_extra - command -v quota-axi >/dev/null 2>&1 || return 1 - if [ -n "$timeout" ]; then - case "$timeout" in - ''|*[!0-9]*|0) return 1 ;; - esac - [ "$(type -t fm_run_timed)" = function ] || return 1 - output=$(fm_run_timed "$timeout" quota-axi --version 2>/dev/null /dev/null /dev/null 2>&1 || return 1 + if [ -n "$timeout" ]; then + case "$timeout" in + ''|*[!0-9]*|0) return 1 ;; + esac + [ "$(type -t fm_run_timed)" = function ] || return 1 + output=$(fm_run_timed "$timeout" quota-axi --version 2>/dev/null /dev/null still name one live process. +fm_remote_job_recorded_owner_alive() { # + local dir=$1 pid recorded_start actual_start recorded_command actual_command + [ -d "$dir" ] && [ ! -L "$dir" ] || return 1 + pid=$(fm_remote_job_read_single_line "$dir/pid" 64 2>/dev/null) || return 1 case "$pid" in ''|*[!0-9]*) return 1 ;; esac [ "$pid" -gt 1 ] || return 1 - recorded_start=$(fm_remote_job_read_single_line "$lock/start" 256) || return 1 + recorded_start=$(fm_remote_job_read_single_line "$dir/start" 256 2>/dev/null) || return 1 actual_start=$(fm_remote_job_process_start "$pid") || return 1 [ "$recorded_start" = "$actual_start" ] || return 1 - recorded_command=$(fm_remote_job_read_single_line "$lock/command" 8192) || return 1 + recorded_command=$(fm_remote_job_read_single_line "$dir/command" 8192 2>/dev/null) || return 1 actual_command=$(fm_remote_job_process_command "$pid") || return 1 [ "$recorded_command" = "$actual_command" ] || return 1 - FM_REMOTE_JOB_OWNER_PID=$pid + FM_REMOTE_JOB_RECORDED_PID=$pid +} + +fm_remote_job_lock_owner_matches_process() { + local account_home=$1 + fm_remote_job_prepare_state "$account_home" || return 1 + fm_remote_job_recorded_owner_alive "$(fm_remote_job_worker_lock_path)" || return 1 + FM_REMOTE_JOB_OWNER_PID=$FM_REMOTE_JOB_RECORDED_PID } fm_remote_job_worker_owned_alive() { @@ -1128,7 +1134,7 @@ fm_remote_job_worker_alive() { # kill -0 "$pid" 2>/dev/null } -fm_remote_job_probe() { # ; a fresh worker heartbeat or active job proves readiness +fm_remote_job_probe() { # ; require a fresh heartbeat outside an active job local account_home=$1 ready lock mtime now [ "${FM_REMOTE_JOB_ACTIVE:-}" = 1 ] && return 0 fm_remote_job_prepare_state "$account_home" || return 1 @@ -1162,7 +1168,7 @@ fm_remote_job_write_launchagent() { # fi [ -d "$FM_REMOTE_JOB_LAUNCH_AGENT_DIR" ] && [ ! -L "$FM_REMOTE_JOB_LAUNCH_AGENT_DIR" ] || return 1 [ -d "$FM_REMOTE_JOB_LAUNCH_AGENT_LOG_DIR" ] && [ ! -L "$FM_REMOTE_JOB_LAUNCH_AGENT_LOG_DIR" ] || return 1 - tmp="$FM_REMOTE_JOB_LAUNCH_AGENT_DIR/.$FM_REMOTE_JOB_LABEL.plist.tmp.$$" + tmp="$FM_REMOTE_JOB_LAUNCH_AGENT_DIR/.$FM_REMOTE_JOB_LABEL.plist.tmp.${BASHPID:-$$}" fm_remote_job_render_launchagent "$root" "$account_home" > "$tmp" || { rm -f -- "$tmp" FM_REMOTE_JOB_ERROR="remote job paths cannot be embedded safely in a property list" @@ -1176,10 +1182,121 @@ fm_remote_job_write_launchagent() { # } } +# The LaunchAgent repair mutex is a symlink naming its holder's record +# directory. Reclaiming a dead holder first renames that uniquely named +# directory into a tomb naming the reclaimer, which elects exactly one +# reclaimer per dead holder, and only then repoints the dangling link. A +# reclaimer that died mid-way leaves its tomb for the next caller to re-elect. +fm_remote_job_reload_lock_take() { # + local lock=$1 from=$2 tomb=$3 name=$4 link + if [ "$from" != "$tomb" ]; then mv -- "$from" "$tomb" 2>/dev/null || return 1; fi + [ ! -e "$lock" ] || return 1 + link="${lock%/*}/$name.link" + rm -f -- "$link" + ln -s "$name" "$link" || return 1 + mv -f -- "$link" "$lock" || { rm -f -- "$link"; return 1; } + rm -rf -- "$tomb" +} + +fm_remote_job_reload_lock_acquire() { # + local lock=$1 dir name owner pid target tomb deadline + dir=${lock%/*} + pid=${BASHPID:-$$} + name="${lock##*/}.owner.$pid.$RANDOM$RANDOM" + owner="$dir/$name" + FM_REMOTE_JOB_RELOAD_OWNER= + (umask 077; mkdir "$owner") 2>/dev/null || return 1 + if ! printf '%s\n' "$pid" > "$owner/pid" || + ! fm_remote_job_process_start "$pid" > "$owner/start" || + ! fm_remote_job_process_command "$pid" > "$owner/command"; then + rm -rf -- "$owner" + return 1 + fi + deadline=$((SECONDS + 120)) + while [ "$SECONDS" -lt "$deadline" ]; do + ln -sn "$name" "$lock" 2>/dev/null && break + if ! target=$(readlink "$lock" 2>/dev/null); then + [ -e "$lock" ] && break + continue + fi + case "$target" in */*) break ;; "${lock##*/}".owner.*) ;; *) break ;; esac + if [ -d "$dir/$target" ] && [ ! -L "$dir/$target" ]; then + if ! fm_remote_job_recorded_owner_alive "$dir/$target"; then + fm_remote_job_reload_lock_take "$lock" "$dir/$target" "$dir/$target.reaped.$name" "$name" && break + fi + else + for tomb in "$dir/$target".reaped.*; do + [ -d "$tomb" ] && [ ! -L "$tomb" ] || continue + if [ "${tomb##*.reaped.}" = "$name" ] || ! fm_remote_job_recorded_owner_alive "$dir/${tomb##*.reaped.}"; then + fm_remote_job_reload_lock_take "$lock" "$tomb" "$dir/$target.reaped.$name" "$name" && break 2 + fi + done + fi + sleep 0.1 + done + if [ "$(readlink "$lock" 2>/dev/null)" = "$name" ]; then + FM_REMOTE_JOB_RELOAD_OWNER=$owner + return 0 + fi + rm -rf -- "$owner" + return 1 +} + +fm_remote_job_reload_lock_release() { # + local lock=$1 owner=${FM_REMOTE_JOB_RELOAD_OWNER:-} status=0 + [ -n "$owner" ] || return 1 + if [ "$(readlink "$lock" 2>/dev/null)" = "${owner##*/}" ]; then + rm -f -- "$lock" || status=1 + else + status=1 + fi + rm -rf -- "$owner" + FM_REMOTE_JOB_RELOAD_OWNER= + return "$status" +} + +# launchd's own record of the process it runs for the agent, so a verified lock +# owner that launchd lost track of is never mistaken for the current worker. +fm_remote_job_launchagent_pid() { # + local root=$1 account_home=$2 uid=$3 pid + fm_remote_job_launchagent_loaded "$root" "$account_home" "$uid" || return 1 + pid=$(launchctl print "gui/$uid/$FM_REMOTE_JOB_LABEL" 2>/dev/null | awk ' + $1 == "pid" && $2 == "=" { print $3; exit } + ') + case "$pid" in ''|*[!0-9]*) return 1 ;; esac + kill -0 "$pid" 2>/dev/null || return 1 + printf '%s\n' "$pid" +} + +fm_remote_job_launchagent_tracks() { # + local tracked + tracked=$(fm_remote_job_launchagent_pid "$1" "$2" "$3") || return 1 + [ "$tracked" = "$4" ] +} + +# The verified lock owner is the launchd-tracked worker and runs current code, +# or has not published its code identity yet. +fm_remote_job_launchagent_owner_current() { # + local root=$1 account_home=$2 uid=$3 identity + fm_remote_job_lock_owner_matches_process "$account_home" || return 1 + fm_remote_job_launchagent_tracks "$root" "$account_home" "$uid" "$FM_REMOTE_JOB_OWNER_PID" || return 1 + identity=$(fm_remote_job_worker_identity_path) + if [ ! -e "$identity" ] && [ ! -L "$identity" ]; then return 0; fi + fm_remote_job_worker_identity_matches "$root" "$account_home" +} + fm_remote_job_reload_launchagent() { # - local account_home=$1 uid=$2 out + local account_home=$1 uid=$2 out i=0 fm_remote_job_launchagent_paths "$account_home" launchctl bootout "gui/$uid/$FM_REMOTE_JOB_LABEL" >/dev/null 2>&1 || true + while launchctl print "gui/$uid/$FM_REMOTE_JOB_LABEL" >/dev/null 2>&1; do + if [ "$i" -ge 100 ]; then + FM_REMOTE_JOB_ERROR="timed out waiting for launchd to finish removing $FM_REMOTE_JOB_LABEL after bootout" + return 1 + fi + i=$((i + 1)) + sleep 0.1 + done if ! out=$(launchctl bootstrap "gui/$uid" "$FM_REMOTE_JOB_LAUNCH_AGENT_PLIST" 2>&1); then FM_REMOTE_JOB_ERROR="launchctl bootstrap gui/$uid refused: ${out:-no diagnostic}" return 1 @@ -1190,6 +1307,74 @@ fm_remote_job_reload_launchagent() { # fi } +fm_remote_job_stale_heartbeat_owner() { # + fm_remote_job_lock_owner_matches_process "$1" || return 1 + printf '%s\n' 'remote-job: ready heartbeat stale while verified worker lock owner is alive' >&2 + FM_REMOTE_JOB_ERROR="remote job worker owns its lock but its ready heartbeat is stale" +} + +# Hold the repair mutex across classification, replacement, and bounded startup +# waits; recompute identity here rather than using a pre-mutex reading that could +# stop another caller's replacement. A launchd-tracked live process gets a startup +# wait even before publishing its lock, including after its repairing caller dies. +# A verified live lock owner after a failed probe wait blocks timeout-driven +# reloads; stale-code and untracked owners take the identity-safe stop path. +fm_remote_job_repair_launchagent() { # + local root=$1 account_home=$2 uid=$3 + if ! fm_remote_job_launchagent_contract_matches "$root" "$account_home"; then + fm_remote_job_write_launchagent "$root" "$account_home" || return 1 + FM_REMOTE_JOB_REPAIRED=1 + fi + if [ "$FM_REMOTE_JOB_REPAIRED" -eq 0 ] && fm_remote_job_launchagent_owner_current "$root" "$account_home" "$uid"; then + fm_remote_job_wait_for_probe "$root" "$account_home" && return 0 + fm_remote_job_stale_heartbeat_owner "$account_home" && return 1 + elif fm_remote_job_lock_owner_matches_process "$account_home"; then + # Only stop the lock owner after the shared pid, start-time, and command + # checks have all verified it as this worker. + fm_remote_job_stop_worker_tree "$FM_REMOTE_JOB_OWNER_PID" || { + FM_REMOTE_JOB_ERROR="stale or untracked remote job worker did not stop safely" + return 1 + } + FM_REMOTE_JOB_REPAIRED=1 + elif [ "$FM_REMOTE_JOB_REPAIRED" -eq 0 ] && fm_remote_job_launchagent_pid "$root" "$account_home" "$uid" >/dev/null; then + fm_remote_job_wait_for_probe "$root" "$account_home" && return 0 + fm_remote_job_stale_heartbeat_owner "$account_home" && return 1 + fi + if [ "$FM_REMOTE_JOB_REPAIRED" -eq 1 ] || + ! fm_remote_job_launchagent_loaded "$root" "$account_home" "$uid" || + ! fm_remote_job_worker_identity_matches "$root" "$account_home"; then + fm_remote_job_reload_launchagent "$account_home" "$uid" || return 1 + FM_REMOTE_JOB_REPAIRED=1 + fi + fm_remote_job_wait_for_probe "$root" "$account_home" && return 0 + fm_remote_job_stale_heartbeat_owner "$account_home" && return 1 + fm_remote_job_reload_launchagent "$account_home" "$uid" || return 1 + FM_REMOTE_JOB_REPAIRED=1 + fm_remote_job_wait_for_probe "$root" "$account_home" && return 0 + # shellcheck disable=SC2034 # Sourceable API consumed by the entrypoint and remote doctor. + FM_REMOTE_JOB_ERROR="remote job worker did not report ready after startup" + return 1 +} + +fm_remote_job_ensure_launchagent() { # + local root=$1 account_home=$2 uid=$3 lock status + fm_remote_job_prepare_state "$account_home" || return 1 + if fm_remote_job_launchagent_contract_matches "$root" "$account_home" && + fm_remote_job_launchagent_owner_current "$root" "$account_home" "$uid" && + fm_remote_job_probe "$account_home" && fm_remote_job_worker_identity_matches "$root" "$account_home"; then + return 0 + fi + lock="$FM_REMOTE_JOB_STATE/launchagent.repair" + fm_remote_job_reload_lock_acquire "$lock" || { + FM_REMOTE_JOB_ERROR="timed out waiting for the remote job LaunchAgent repair lock" + return 1 + } + fm_remote_job_repair_launchagent "$root" "$account_home" "$uid" + status=$? + fm_remote_job_reload_lock_release "$lock" || true + return "$status" +} + fm_remote_job_start_linux_worker() { # local root=$1 account_home=$2 worker pid worker="$root/bin/fm-remote-job-worker.sh" @@ -1228,7 +1413,7 @@ fm_remote_job_start_linux_worker() { # } fm_remote_job_ensure_worker() { # - local root=$1 account_home=$2 platform uid identity_matches=0 + local root=$1 account_home=$2 platform uid FM_REMOTE_JOB_ERROR= FM_REMOTE_JOB_REPAIRED=0 root=$(fm_remote_job_canonical_existing_dir "$root") || { @@ -1245,7 +1430,6 @@ fm_remote_job_ensure_worker() { # return 1 } platform=$(fm_remote_job_platform) - fm_remote_job_worker_identity_matches "$root" "$account_home" && identity_matches=1 if [ "$platform" = darwin ]; then uid=$(id -u 2>/dev/null || true) case "$uid" in ''|*[!0-9]*) FM_REMOTE_JOB_ERROR="remote account uid is unavailable; run fm-on.sh fm-remote-doctor.sh --fix"; return 1 ;; esac @@ -1253,32 +1437,17 @@ fm_remote_job_ensure_worker() { # FM_REMOTE_JOB_ERROR="no Aqua login session exists for uid $uid; log that account in at the console, then run fm-on.sh fm-remote-doctor.sh --fix" return 1 fi - if ! fm_remote_job_launchagent_contract_matches "$root" "$account_home"; then - fm_remote_job_write_launchagent "$root" "$account_home" || return 1 - FM_REMOTE_JOB_REPAIRED=1 - fi - if ! fm_remote_job_launchagent_loaded "$root" "$account_home" "$uid" || - [ "$FM_REMOTE_JOB_REPAIRED" -eq 1 ] || [ "$identity_matches" -eq 0 ]; then - fm_remote_job_reload_launchagent "$account_home" "$uid" || return 1 - FM_REMOTE_JOB_REPAIRED=1 - fi - else - fm_remote_job_start_linux_worker "$root" "$account_home" || return 1 + fm_remote_job_ensure_launchagent "$root" "$account_home" "$uid" + return fi + fm_remote_job_start_linux_worker "$root" "$account_home" || return 1 + fm_remote_job_wait_for_probe "$root" "$account_home" && return 0 + # A replaced Linux supervisor can lose its first ownership race while the + # prior supervisor finishes releasing the shared worker lock. Retry the + # idempotent start once before reporting a startup failure. + fm_remote_job_start_linux_worker "$root" "$account_home" || return 1 + FM_REMOTE_JOB_REPAIRED=1 fm_remote_job_wait_for_probe "$root" "$account_home" && return 0 - if [ "$platform" = darwin ]; then - fm_remote_job_reload_launchagent "$account_home" "$uid" || return 1 - FM_REMOTE_JOB_REPAIRED=1 - fm_remote_job_wait_for_probe "$root" "$account_home" && return 0 - else - # A replaced Linux supervisor can lose its first ownership race while the - # prior supervisor finishes releasing the shared worker lock. Retry the - # idempotent start once, matching the bounded recovery already used above - # for launchd, before reporting a startup failure. - fm_remote_job_start_linux_worker "$root" "$account_home" || return 1 - FM_REMOTE_JOB_REPAIRED=1 - fm_remote_job_wait_for_probe "$root" "$account_home" && return 0 - fi # shellcheck disable=SC2034 # Sourceable API consumed by the entrypoint and remote doctor. FM_REMOTE_JOB_ERROR="remote job worker did not report ready after startup" return 1 diff --git a/bin/fm-remote-job-worker.sh b/bin/fm-remote-job-worker.sh index 118d75c3493..5cdbd72873b 100755 --- a/bin/fm-remote-job-worker.sh +++ b/bin/fm-remote-job-worker.sh @@ -28,9 +28,14 @@ # one second between passes. Work arriving after the four-pass burst may wait # for that quiet scan. Newly staged or cancelled work, a lane that died, an # orphaned claim, or an expired queue deadline can wait that interval plus -# scan work and scheduling time. It refreshes the readiness heartbeat about once -# per second, far inside the probe's 10-second freshness bound. The stale -# sweep, whose state preparation also re-applies the queue directories' 0700 +# scan work and scheduling time. A separate heartbeat process refreshes readiness +# about once per second, including during slow scans and sweeps, only while the +# serving process is alive and its recorded lock ownership still verifies. +# The heartbeat recreates a missing ready file with the serving process's PID +# and mode 0600 after verifying ownership, without waiting for the serving loop. +# Losing lock ownership stops heartbeat refresh; losing the heartbeat process +# while still owning the lock stops the serving loop on its next pass. +# The stale sweep, whose state preparation also re-applies the queue directories' 0700 # modes, runs at startup and then at most every 60 seconds, never more rarely # than the shortest record reap age. # @@ -77,6 +82,7 @@ WORKER_LOCK_HELD=0 WORKER_LOCK_BOUND= WORKER_RELEASE_OWNERSHIP=1 WORKER_SUPERVISED_PID= +WORKER_HEARTBEAT_PID= WORKER_PREEMPTIBLE=0 WORKER_PREEMPTED=0 WORKER_LANE_HOME= @@ -97,15 +103,47 @@ worker_account_home() { CDPATH='' cd ~ 2>/dev/null && pwd -P } -worker_write_heartbeat() { - local ready tmp +worker_write_heartbeat() { # + local owner=$1 ready tmp ready=$(fm_remote_job_worker_ready_path) tmp=$(umask 077; mktemp "$FM_REMOTE_JOB_STATE/.ready.XXXXXX") || return 1 - printf '%s\n' "${BASHPID:-$$}" > "$tmp" || { rm -f -- "$tmp"; return 1; } + printf '%s\n' "$owner" > "$tmp" || { rm -f -- "$tmp"; return 1; } chmod 600 "$tmp" || { rm -f -- "$tmp"; return 1; } mv -f -- "$tmp" "$ready" } +worker_heartbeat_loop() { # + local account_home=$1 owner=$2 ready owner_state + ready=$(fm_remote_job_worker_ready_path) + trap 'exit 0' HUP INT TERM + while kill -0 "$owner" 2>/dev/null && + owner_state=$(/bin/ps -p "$owner" -o state= 2>/dev/null) && + [ -n "$owner_state" ] && [[ "$owner_state" != *Z* ]] && + fm_remote_job_lock_owner_matches_process "$account_home" && + [ "$FM_REMOTE_JOB_OWNER_PID" = "$owner" ]; do + if [ ! -e "$ready" ] && [ ! -L "$ready" ]; then + worker_write_heartbeat "$owner" || exit 1 + else + touch -c -- "$ready" || exit 1 + fi + /bin/sleep 1 + done +} + +worker_start_heartbeat() { # + local account_home=$1 owner=${BASHPID:-$$} + worker_heartbeat_loop "$account_home" "$owner" & + WORKER_HEARTBEAT_PID=$! +} + +worker_stop_heartbeat() { + local pid=${WORKER_HEARTBEAT_PID:-} + [ -n "$pid" ] || return 0 + kill -TERM "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + WORKER_HEARTBEAT_PID= +} + worker_publish_pid() { local pid_file tmp pid_file=$(fm_remote_job_worker_pid_path) @@ -115,9 +153,8 @@ worker_publish_pid() { mv -f -- "$tmp" "$pid_file" } -worker_publish_identity() { - local account_home=$1 identity identity_file tmp - identity=$(fm_remote_job_code_identity "$FM_ROOT" "$account_home") || return 1 +worker_publish_identity() { # + local identity=$1 identity_file tmp identity_file=$(fm_remote_job_worker_identity_path) tmp=$(umask 077; mktemp "$FM_REMOTE_JOB_STATE/.identity.XXXXXX") || return 1 printf '%s\n' "$identity" > "$tmp" || { rm -f -- "$tmp"; return 1; } @@ -174,12 +211,16 @@ worker_recover_quarantine() { # rm -f -- "$WORKER_LOCK/quarantine" } -worker_acquire_lock() { - local account_home=$1 attempt=0 +worker_acquire_lock() { # + local account_home=$1 identity=$2 attempt=0 while [ "$attempt" -lt 150 ]; do if (umask 077; mkdir "$WORKER_LOCK") 2>/dev/null; then WORKER_LOCK_HELD=1 + # Discard the predecessor's heartbeat before publishing our identity: + # readiness must come from this owner after lock publication succeeds. + rm -f -- "$(fm_remote_job_worker_ready_path)" || return 1 worker_publish_lock_owner || return 1 + worker_publish_identity "$identity" || return 4 return 0 fi [ -d "$WORKER_LOCK" ] && [ ! -L "$WORKER_LOCK" ] || return 1 @@ -533,6 +574,7 @@ worker_shutdown() { } worker_exit_cleanup() { + worker_stop_heartbeat if [ "$WORKER_RELEASE_OWNERSHIP" -eq 1 ] && ! worker_stop_active_execution; then worker_error "could not stop the active command tree during exit" worker_publish_quarantine || worker_error "could not quarantine failed exit ownership" @@ -1158,24 +1200,27 @@ worker_wait_for_work() { } main() { - local account_home lock_status next_heartbeat=-1 next_sweep=0 sweep_interval + local account_home identity lock_status next_sweep=0 sweep_interval account_home=$(worker_account_home) || { worker_error "cannot resolve account home"; exit 1; } FM_ROOT=$(fm_remote_job_canonical_existing_dir "$FM_ROOT") || { worker_error "configured FM_ROOT is unsafe"; exit 1; } [ -f "$FM_ROOT/AGENTS.md" ] && [ ! -L "$FM_ROOT/AGENTS.md" ] || { worker_error "FM_ROOT is not a Firstmate checkout"; exit 1; } fm_remote_job_prepare_state "$account_home" || { worker_error "$FM_REMOTE_JOB_ERROR"; exit 1; } + identity=$(fm_remote_job_code_identity "$FM_ROOT" "$account_home") || { worker_error "cannot compute worker code identity"; exit 1; } WORKER_LOCK=$(fm_remote_job_worker_lock_path) trap worker_exit_cleanup EXIT - worker_acquire_lock "$account_home" + worker_acquire_lock "$account_home" "$identity" lock_status=$? case "$lock_status" in 0) ;; 2) exit 0 ;; 3) worker_error "worker ownership is quarantined after an unconfirmed shutdown"; exit 75 ;; + 4) worker_error "cannot publish worker code identity"; exit 1 ;; *) worker_error "cannot acquire or safely reclaim worker ownership"; exit 1 ;; esac trap worker_shutdown HUP INT TERM - worker_publish_identity "$account_home" || { worker_error "cannot publish worker code identity"; exit 1; } worker_publish_pid || { worker_error "cannot publish worker pid"; exit 1; } + worker_write_heartbeat "${BASHPID:-$$}" || { worker_error "cannot update worker heartbeat"; exit 1; } + worker_start_heartbeat "$account_home" sweep_interval=$WORKER_SWEEP_SECONDS [ "$FM_REMOTE_JOB_STAGE_REAP_SECONDS" -ge "$sweep_interval" ] || sweep_interval=$FM_REMOTE_JOB_STAGE_REAP_SECONDS [ "$FM_REMOTE_JOB_REAP_SECONDS" -ge "$sweep_interval" ] || sweep_interval=$FM_REMOTE_JOB_REAP_SECONDS @@ -1183,13 +1228,14 @@ main() { WORKER_FAST_REMAINING=0 WORKER_ACTIVITY=1 while :; do - if [ "$SECONDS" -ne "$next_heartbeat" ]; then - worker_write_heartbeat || { worker_error "cannot update worker heartbeat"; exit 1; } - next_heartbeat=$SECONDS + # The independent heartbeat keeps readiness fresh through slow serving + # passes while ownership remains verifiable. + if ! kill -0 "$WORKER_HEARTBEAT_PID" 2>/dev/null && + fm_remote_job_lock_owner_matches_process "$account_home" && + [ "$FM_REMOTE_JOB_OWNER_PID" = "${BASHPID:-$$}" ]; then + worker_error "readiness heartbeat process stopped" + exit 1 fi - # Checked right after a heartbeat no older than a second, so the grace - # window cannot make a still-healthy worker read as unready to a - # concurrent probe. if worker_code_root_abandoned; then worker_error "configured FM_ROOT $FM_ROOT no longer exists; stopping the abandoned worker" exit 0 diff --git a/bin/fm-send.sh b/bin/fm-send.sh index 19562680313..cdd4ddcb95d 100755 --- a/bin/fm-send.sh +++ b/bin/fm-send.sh @@ -47,15 +47,10 @@ # instruction. There is no delivered-unconfirmed # outcome on this plane: "did the doorbell land" is no longer the question - # "was the message acted on" is, and that is answered asynchronously for an -# ordinary record by the worker's acknowledgement move into handled/. The -# watcher re-rings an unacknowledged message while its endpoint remains -# available, escalates after the bounded ladder, and instead routes a positively -# dead or missing endpoint directly to recovery without typing. An explicit -# fire-and-forget record is excluded from that ladder; when config/wait-no-turns -# is present and its ring here was skipped or failed, the watcher rings it -# exactly once more. -# bin/fm-task-inbox-lib.sh owns the record format, the doorbell line, and the -# re-ring ladder. The composer pre-check before the ring is ADVISORY only: when +# ordinary record by the worker's acknowledgement move into handled/. +# bin/fm-task-inbox-lib.sh owns the record format, doorbell line, and retry and +# escalation policy for ordinary and fire-and-forget records. +# The composer pre-check before the ring is ADVISORY only: when # the composer visibly holds pending text the ring is skipped with a notice and # the watcher re-rings an ordinary record later; no composer verdict is # delivery proof on this plane, and a failed ring never fails the send. diff --git a/bin/fm-supervise-daemon.sh b/bin/fm-supervise-daemon.sh index a2a7664f4fb..1364203efb5 100755 --- a/bin/fm-supervise-daemon.sh +++ b/bin/fm-supervise-daemon.sh @@ -7,9 +7,9 @@ # ESCALATES a batched, distilled digest to the supervisor pane on # captain-relevant events plus bounded declared-wait rechecks. This is the # token-efficient replacement for the prior always-inject daemon: routine -# signal/stale/heartbeat wakes cost zero firstmate context; only done/ -# needs-decision/blocked/failed/persistent-wedge/check-output events and a -# declared-wait recheck reach the LLM, and even then as one pre-read digest per +# signal/stale/heartbeat wakes cost zero firstmate context; routing is owned by +# .agents/skills/afk/SKILL.md (Classification policy). +# Escalated events reach the LLM as one pre-read digest per # batch window. That digest is byte-bounded (see escalate_flush); when it cuts # or omits anything it names a state/.subsuper-digests/ file holding every # buffered event verbatim. @@ -48,7 +48,7 @@ # drain and acknowledges it only after routing completes. # - Fail-safe-to-escalate: any wake the classifier cannot confidently mark # routine is escalated. -# - Bounded wedge latency: a stale pane without a declared wait is escalated +# - Bounded wedge latency: ordinary pane staleness without a declared wait escalates # only after it has been idle for STALE_ESCALATE_SECS # (configurable), rechecked once. A wedged crewmate is therefore detected # within STALE_ESCALATE_SECS + a tick, never lost. A declared wait - either a @@ -1562,6 +1562,8 @@ handle_wake() { # *) arg="${reason#signal: }" ;; esac decision=$(FM_STATUS_SPAN_ENDPOINT_FILE="$capture" classify_signal "$arg" "$state") ;; + stale:*" (unread firstmate instruction: stuck-busy "*|stale:*" (steering-inbox busy bookkeeping unwritable: "*) + decision="escalate|${reason#stale: }" ;; stale:*) kind=stale; arg="${reason#stale: }"; stale_detail="${arg#"$arg"}" case "$arg" in *" ("*) stale_detail="${arg#*" ("}"; arg="${arg%% \(*}" ;; esac task=$(window_to_task "$arg" "$state") diff --git a/bin/fm-task-inbox-lib.sh b/bin/fm-task-inbox-lib.sh index 257da1dd54d..3b87a429a21 100644 --- a/bin/fm-task-inbox-lib.sh +++ b/bin/fm-task-inbox-lib.sh @@ -28,6 +28,7 @@ # .inbox/.seq.lock serializes sequence allocation across writers # (the session and the away daemon) # .inbox/.ring-state watcher re-ring ladder: "\t\t" +# .inbox/.busy-state consecutive busy deferrals: "\t" # .inbox/.escalated oldest-message name already surfaced as stale, # so later polls suppress another escalation # .inbox/.retry-ring name of a fire-and-forget record still owed its @@ -52,10 +53,14 @@ # attempt may ring or be skipped to protect another draft in a proven pending # composer; an unsubmitted copy of this doorbell is retried. After # FM_TASK_INBOX_RING_MAX attempts without an acknowledgement it escalates. The -# caller owns the busy and recovery-grade endpoint checks: a busy pane waits, -# while a positively dead or missing endpoint skips delivery and the ladder and -# escalates directly. This library owns only the schedule and escalation marker. -# If attempt bookkeeping cannot be persisted while the record remains unhandled, +# caller owns the busy and recovery-grade endpoint checks: due actions deferred +# by a busy pane consume a separate durable consecutive-poll budget, +# FM_TASK_INBOX_BUSY_MAX. At that bound the same escalation path surfaces a +# stuck-busy reason without typing. A non-busy due check or acknowledgement resets +# this budget. Fire-and-forget retries remain outside escalation. A positively +# dead or missing endpoint skips delivery and the ladder and escalates directly. +# This library owns the schedule, durable budgets, and escalation marker. +# If delivery-attempt or busy-deferral bookkeeping fails while the record remains unhandled, # the caller surfaces that failure instead of retrying silently; a concurrently # removed inbox is a quiet no-op. Escalation deliberately queues the wake before # writing the deduplication marker: normal polls surface a message once, while a @@ -85,6 +90,7 @@ # Tunables (env): # FM_TASK_INBOX_GRACE_SECS default 90; delivery-attempt grace and spacing # FM_TASK_INBOX_RING_MAX default 3; delivery attempts before escalation +# FM_TASK_INBOX_BUSY_MAX default 2; consecutive busy-deferred due polls before escalation _FM_TASK_INBOX_LIB_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" # Both dependencies are canonical lint roots in their own right. Keep them as @@ -98,6 +104,7 @@ _FM_TASK_INBOX_LIB_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" FM_TASK_INBOX_SCHEMA='fm-task-inbox.v1' FM_TASK_INBOX_GRACE_DEFAULT=90 FM_TASK_INBOX_RING_MAX_DEFAULT=3 +FM_TASK_INBOX_BUSY_MAX_DEFAULT=2 FM_TASK_INBOX_LOCK_WAIT_DEFAULT=5 fm_task_inbox_grace_secs() { @@ -112,6 +119,40 @@ fm_task_inbox_ring_max() { printf '%s' "$m" } +fm_task_inbox_busy_max() { + local m=${FM_TASK_INBOX_BUSY_MAX:-$FM_TASK_INBOX_BUSY_MAX_DEFAULT} + case "$m" in ''|*[!0-9]*) m=$FM_TASK_INBOX_BUSY_MAX_DEFAULT ;; esac + # Check the length before numeric comparison so oversized input cannot overflow. + if [ "${#m}" -gt 9 ] || [ "$m" -eq 0 ]; then + m=$FM_TASK_INBOX_BUSY_MAX_DEFAULT + fi + printf '%s' "$m" +} + +# Persist before returning the new count, so a fresh watcher continues the same +# bounded wait. A removed or acknowledged record is a quiet no-op. +fm_task_inbox_record_busy() { # + local dir base previous count + dir=$(fm_task_inbox_dir "$1" "$2") + base=${3##*/} + { IFS=$(printf '\t') read -r previous count < "$dir/.busy-state"; } 2>/dev/null || true + [ "${previous:-}" = "$base" ] || count=0 + case "${count:-}" in ''|*[!0-9]*) count=0 ;; esac + [ -f "$3" ] || { printf '0'; return 0; } + count=$((count + 1)) + if ! { printf '%s\t%s\n' "$base" "$count" > "$dir/.busy-state"; } 2>/dev/null; then + [ -f "$3" ] || { printf '0'; return 0; } + return 1 + fi + printf '%s' "$count" +} + +fm_task_inbox_clear_busy() { # + local dir + dir=$(fm_task_inbox_dir "$1" "$2") + rm -f "$dir/.busy-state" 2>/dev/null +} + fm_task_inbox_dir() { # printf '%s/%s.inbox' "$1" "$2" } @@ -415,7 +456,7 @@ fm_task_inbox_due_action() { # local dir oldest base now grace max ladder rec_base count last dir=$(fm_task_inbox_dir "$1" "$2") if ! oldest=$(fm_task_inbox_oldest_unhandled "$1" "$2"); then - rm -f "$dir/.ring-state" "$dir/.escalated" 2>/dev/null || true + rm -f "$dir/.ring-state" "$dir/.escalated" "$dir/.busy-state" 2>/dev/null || true # The one retry ring exists only while config/wait-no-turns is present. # Absent, a mark is left untouched and the inbox stays quiet, as before. if [ -e "${FM_CONFIG_OVERRIDE:-${FM_HOME:-}/config}/wait-no-turns" ]; then @@ -443,13 +484,8 @@ fm_task_inbox_due_action() { # $ladder EOF if [ -n "$rec_base" ] && [ "$rec_base" != "$base" ]; then - # A different oldest message: the previous ladder is stale. An absent - # ladder is left alone so a dead-pane escalation, which never rings and so - # never writes one, keeps its marker (the marker check below still ignores - # a marker naming some other message). count=0 last=0 - rm -f "$dir/.escalated" 2>/dev/null || true fi case "$count" in ''|*[!0-9]*) count=0 ;; esac case "$last" in ''|*[!0-9]*) last=0 ;; esac diff --git a/bin/fm-teardown.sh b/bin/fm-teardown.sh index 61cc21d08f9..d00ba1e48c6 100755 --- a/bin/fm-teardown.sh +++ b/bin/fm-teardown.sh @@ -65,7 +65,8 @@ # by itself causes a false refusal of landed work. # A gh lookup error falls back to the content check; if that is also inconclusive, # teardown refuses rather than risk discarding unlanded work. -# Uncommitted changes are never landed. +# Uncommitted changes are never landed; dirty refusals distinguish untracked-only +# leftovers from tracked edits and list at most ten non-exempt untracked paths. # local-only projects additionally accept work merged into the local default # branch (firstmate performs that merge after configured approval) as a fallback # for the common case where there is no remote at all. @@ -1873,6 +1874,23 @@ teardown_treehouse_return() { return 1 } +report_worktree_dirt() { + # Use the same porcelain snapshot and exemptions as the refusal predicate. + printf '%s\n' "$1" | awk ' + /^\?\? / { if (++untracked <= 10) paths = paths " " substr($0, 4) "\n"; next } + NF { tracked = 1 } + END { + if (tracked) print "uncommitted changes present (includes tracked edits)" + else print "uncommitted changes present (untracked-only leftovers)" + if (untracked) { + print "untracked paths (up to 10):" + printf "%s", paths + if (untracked > 10) print " ... additional untracked paths omitted" + } + } + ' >&2 +} + validate_worktree_teardown_safety() { local dirty_raw dirty unpushed_raw unpushed DEFAULT unmerged_raw unmerged branch [ -d "$WT" ] || return 0 @@ -1889,7 +1907,7 @@ validate_worktree_teardown_safety() { echo "Restore the git index state, or get the captain's explicit OK to discard, then --force." >&2 return 1 fi - dirty=$(printf '%s\n' "$dirty_raw" | grep -vE '^\?\? (\.claude/|\.fm-(grok|kimi)-turnend$)' | head -1 || true) + dirty=$(printf '%s\n' "$dirty_raw" | grep -vE '^\?\? (\.claude/|\.fm-(grok|kimi)-turnend$)' || true) if ! unpushed_raw=$(git -C "$WT" log --oneline HEAD --not --remotes -- 2>/dev/null); then if worktree_safety_blocked_by_lock "commits not on a remote"; then @@ -1914,14 +1932,14 @@ validate_worktree_teardown_safety() { unmerged=$(printf '%s\n' "$unmerged_raw" | head -5) if [ -n "$dirty" ] || [ -n "$unmerged" ]; then echo "REFUSED: local-only worktree $WT has work not yet merged into $DEFAULT and not on any remote." >&2 - [ -n "$dirty" ] && echo "uncommitted changes present" >&2 + [ -n "$dirty" ] && report_worktree_dirt "$dirty" [ -n "$unmerged" ] && printf 'commits not yet on %s:\n%s\n' "$DEFAULT" "$unmerged" >&2 echo "Merge the branch into local $DEFAULT first (bin/fm-merge-local.sh after the captain approves), or push to a fork/remote, or get the captain's explicit OK to discard, then --force." >&2 return 1 fi elif [ -n "$dirty" ]; then echo "REFUSED: worktree $WT has uncommitted changes." >&2 - echo "uncommitted changes present" >&2 + report_worktree_dirt "$dirty" echo "Commit them (or get the captain's explicit OK to discard, then --force)." >&2 return 1 elif [ -n "$unpushed" ]; then diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 9ccac5faabd..90d2602c678 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -77,12 +77,10 @@ # interrupt, signal, or restart of the worker or its # tool process. # stale: (unread firstmate instruction: ...) -# the steering-inbox ladder spent its delivery-attempt -# budget on an idle pane without an acknowledgement # stale: (steering-inbox ladder bookkeeping unwritable: ...) -# an unhandled record's ladder cannot advance; quiet -# successful attempts never wake firstmate -# (bin/fm-task-inbox-lib.sh owns the ladder policy) +# stale: (steering-inbox busy bookkeeping unwritable: ...) +# steering-inbox recovery; bin/fm-task-inbox-lib.sh owns +# delivery-attempt, busy-deferral, and unavailable-endpoint policy # check: