diff --git a/.agents/skills/process-event-sources/SKILL.md b/.agents/skills/process-event-sources/SKILL.md index f374f8b2557..4b2629113a6 100644 --- a/.agents/skills/process-event-sources/SKILL.md +++ b/.agents/skills/process-event-sources/SKILL.md @@ -118,13 +118,15 @@ Supported by tests: - the handled acknowledgement is generation-keyed to the exact source and sequence, private, path-safe, durable, and idempotent, and is the only thing that stops re-announcement; - one identity-matched owner per canonical source, across homes that share one underlying source store; - registration and ownership transitions share one per-source boundary, release is generation-bound, and uncertain process identity preserves the source for retry; -- ownership moves only when the owner is stale and an independent process-group check proves the whole generation gone, so neither a crashed leader nor a reused pid relaxes cleanup while the old group survives; a safely identified surviving group is stopped before replacement, and the claim is kept for retry when it cannot be; +- leaderless PID/PGID-reuse ambiguity preserves the claim without signalling or replacement, as owned by the operating contract in [`docs/configuration.md`](../../../docs/configuration.md#process-to-event-sources-stateprocevent); +- runner lifetime, owner-lease, and launch-pacing guarantees follow the operating contract in [`docs/configuration.md`](../../../docs/configuration.md#process-to-event-sources-stateprocevent); - stored argv is executed directly, so an argument containing spaces or shell metacharacters is never re-split or interpreted; - oversized output is bounded rather than published whole or silently dropped. The `when` adapter's guarantees are part of the operating contract in [`docs/configuration.md`](../../../docs/configuration.md#process-to-event-sources-stateprocevent). **Not true, and never to be claimed:** at-least-once, no-loss, or lossless delivery, and no generic exactly-once effect either - the handled acknowledgement only stops re-announcement, it says nothing about whether a paired external effect performed before the acknowledgement call actually completed, so a crash between that effect and the call can still repeat the effect on the next replay. +Also never claim that a source cannot refresh its owning home's lease: that rule is confused-agent-grade and a deliberately marker-stripping source is out of scope, per the operating contract in [`docs/configuration.md`](../../../docs/configuration.md#process-to-event-sources-stateprocevent). The currently published `lavish-axi poll` destructively clears feedback before returning it. A result lost after that clearing and before the runner reads the process output is unrecoverable, and no firstmate wrapper can close that source-side window. diff --git a/.pi/extensions/fm-branch-supervision.ts b/.pi/extensions/fm-branch-supervision.ts index c456f00f5a2..d9cef8d7a43 100644 --- a/.pi/extensions/fm-branch-supervision.ts +++ b/.pi/extensions/fm-branch-supervision.ts @@ -78,6 +78,7 @@ import { getAgentDir, keyHint, ModelRuntime, + type ModelRegistry, SessionManager, ToolExecutionComponent, type AgentSession, @@ -644,6 +645,11 @@ export default function (pi: ExtensionAPI) { // extension plus its model_select event, because createBranch runs at wake // time with no context of its own. It is what "follow main" applies. let mainModel: { provider: string; id: string } | null = null; + // Main's own model registry, captured from the contexts Pi hands this + // extension the same way mainModel is. It is the ONLY read path to + // providers an extension registered at runtime (pi-devin-auth's "devin"), + // which the branch's isolated ModelRuntime cannot see on its own. + let mainModelRegistry: ModelRegistry | null = null; // Main's own current effort needs no such tracking: Pi answers it directly // on demand, including at wake time. It throws only when the extension @@ -657,8 +663,9 @@ export default function (pi: ExtensionAPI) { } } - function rememberMainModel(ctx?: { model?: { provider: string; id: string } }): void { + function rememberMainModel(ctx?: { model?: { provider: string; id: string }; modelRegistry?: ModelRegistry }): void { if (ctx?.model) mainModel = { provider: ctx.model.provider, id: ctx.model.id }; + if (ctx?.modelRegistry) mainModelRegistry = ctx.modelRegistry; } function deliverBranchHealthNote(text: string): void { @@ -708,10 +715,55 @@ export default function (pi: ExtensionAPI) { // and same user as main, so stored credentials keep their own semantics // (OAuth stays OAuth, an API key stays an API key) and nothing is ever // installed, converted, derived, or overwritten here. + // A provider that exists only because an extension registered it into + // main's runtime (pi-devin-auth's "devin", whose streamSimple is the custom + // gRPC path no static catalog can express) is invisible to an isolated + // branch runtime until its registration is copied across. The config object + // carries that streamSimple and oauth wiring by reference, so copying it + // reuses the provider's own registration rather than reimplementing its + // wire protocol; the copy is never persisted and stays scoped to this one + // runtime. One registration that fails to compose must not blind the rest, + // so each copy is isolated. A just-registered provider's auth check has not + // run yet, so the copied providers are refreshed here and every caller's + // hasConfiguredAuth verdict is real rather than the provisional entry + // registration leaves behind. + async function copyExtensionProviders(modelRuntime: ModelRuntime): Promise { + if (!mainModelRegistry) return; + let providerIds: readonly string[]; + try { + providerIds = mainModelRegistry.getRegisteredProviderIds(); + } catch { + return; + } + const copied: string[] = []; + for (const providerId of providerIds) { + try { + const config = mainModelRegistry.getRegisteredProviderConfig(providerId); + if (config) { + modelRuntime.registerProvider(providerId, config); + copied.push(providerId); + } + } catch { + // A registration that fails to compose in the isolated runtime leaves + // that provider unavailable, exactly as if it were never copied. + } + } + if (copied.length === 0) return; + try { + await modelRuntime.refresh({ providers: copied, allowNetwork: false }); + } catch { + // A failed availability refresh is answered by hasConfiguredAuth. + } + } + async function resolveBranchModel(provider: string, modelId: string): Promise { const label = `${provider}/${modelId}`; const modelRuntime = await ModelRuntime.create(); - const model = modelRuntime.getModel(provider, modelId) as BranchModel | undefined; + let model = modelRuntime.getModel(provider, modelId) as BranchModel | undefined; + if (!model) { + await copyExtensionProviders(modelRuntime); + model = modelRuntime.getModel(provider, modelId) as BranchModel | undefined; + } if (!model) return { ok: false, reason: `${label} is unavailable to the isolated branch runtime` }; if (!modelRuntime.hasConfiguredAuth(provider)) { return { ok: false, reason: `${label} has no configured credentials in the isolated branch runtime` }; @@ -1659,6 +1711,7 @@ ${context.command} let available: string[]; try { const modelRuntime = await ModelRuntime.create(); + await copyExtensionProviders(modelRuntime); available = ctx.modelRegistry .getAvailable() .filter((model) => modelRuntime.getModel(model.provider, model.id) && modelRuntime.hasConfiguredAuth(model.provider)) diff --git a/bin/backends/herdr.sh b/bin/backends/herdr.sh index 8728b356cc0..55e73d61fdc 100644 --- a/bin/backends/herdr.sh +++ b/bin/backends/herdr.sh @@ -661,7 +661,7 @@ fm_backend_herdr_presentation_lock_namespace() { fm_backend_herdr_presentation_lock_namespace_mode() { if [ "$(uname -s 2>/dev/null)" = Darwin ]; then - stat -f '%Lp' "$1" 2>/dev/null + /usr/bin/stat -f '%Lp' "$1" 2>/dev/null else stat -c '%a' "$1" 2>/dev/null fi @@ -669,7 +669,7 @@ fm_backend_herdr_presentation_lock_namespace_mode() { fm_backend_herdr_presentation_lock_namespace_uid() { if [ "$(uname -s 2>/dev/null)" = Darwin ]; then - stat -f '%u' "$1" 2>/dev/null + /usr/bin/stat -f '%u' "$1" 2>/dev/null else stat -c '%u' "$1" 2>/dev/null fi diff --git a/bin/fm-backlog-receive.sh b/bin/fm-backlog-receive.sh index 15d9bde99ae..46cf06783fd 100755 --- a/bin/fm-backlog-receive.sh +++ b/bin/fm-backlog-receive.sh @@ -57,7 +57,7 @@ list_keys() { # lock_age() { local modified now if [ "$(uname 2>/dev/null)" = Darwin ]; then - modified=$(stat -f '%m' "$1" 2>/dev/null) || return 1 + modified=$(/usr/bin/stat -f '%m' "$1" 2>/dev/null) || return 1 else modified=$(stat -c '%Y' "$1" 2>/dev/null) || return 1 fi diff --git a/bin/fm-bootstrap.sh b/bin/fm-bootstrap.sh index eb015e1d694..1acb9fa8a61 100755 --- a/bin/fm-bootstrap.sh +++ b/bin/fm-bootstrap.sh @@ -938,14 +938,14 @@ x_mode_write_if_changed() { [ "$parent" != "$dest" ] || return 1 [ -d "$parent" ] && [ ! -L "$parent" ] || return 1 if [ "$(uname)" = Darwin ]; then - parent_device=$(stat -f %d "$parent" 2>/dev/null) || return 1 + parent_device=$(/usr/bin/stat -f %d "$parent" 2>/dev/null) || return 1 else parent_device=$(stat -c %d "$parent" 2>/dev/null) || return 1 fi if [ -e "$dest" ] || [ -L "$dest" ]; then fmx_single_link_file_valid "$dest" "$parent_device" || return 1 if [ "$(uname)" = Darwin ]; then - current_mode=$(stat -f %Lp "$dest" 2>/dev/null) || return 1 + current_mode=$(/usr/bin/stat -f %Lp "$dest" 2>/dev/null) || return 1 else current_mode=$(stat -c %a "$dest" 2>/dev/null) || return 1 fi diff --git a/bin/fm-busy-event.sh b/bin/fm-busy-event.sh index 0abcab8ee39..17fde457810 100755 --- a/bin/fm-busy-event.sh +++ b/bin/fm-busy-event.sh @@ -103,7 +103,7 @@ LOCK="$REC.lock" # gets a non-numeric token. Detect the platform once and pick the right form, # exactly as bin/fm-watch.sh does. if [ "$(uname)" = Darwin ]; then - lock_mtime() { stat -f %m "$1" 2>/dev/null; } + lock_mtime() { /usr/bin/stat -f %m "$1" 2>/dev/null; } else lock_mtime() { stat -c %Y "$1" 2>/dev/null; } fi diff --git a/bin/fm-captain-hold.sh b/bin/fm-captain-hold.sh index 2f41dc40b18..3ebc13cb923 100755 --- a/bin/fm-captain-hold.sh +++ b/bin/fm-captain-hold.sh @@ -30,7 +30,7 @@ # fm-captain-hold.sh binding # fm-captain-hold.sh complete (--none | ...) # fm-captain-hold.sh verify -# fm-captain-hold.sh open +# fm-captain-hold.sh open [--identity] [--distinguish-absent] # fm-captain-hold.sh diverged # fm-captain-hold.sh reconcile list # fm-captain-hold.sh reconcile close --evidence-file @@ -157,12 +157,23 @@ # is (not Done, hold kind captain), 1 means it is not, and 2 means the answer # could not be established, so a caller that must never close a live call can # treat "cannot tell" as its own case instead of as a no. With -# `--distinguish-absent`, an absent local task returns 3 instead of 1. It prints -# nothing on these predicate results and mutates nothing. bin/fm-teardown.sh asks it before its automatic +# `--distinguish-absent`, an absent local task returns 3 instead of 1. +# It prints nothing on these predicate results and mutates nothing, unless +# `--identity` asks it to print this call's +# LIFECYCLE identity, which it does on an exit 0 only. That identity - the +# hold-set stamp and the count of recorded answers - is what distinguishes two +# successive calls on one task id: re-holding released work starts a new +# lifecycle without necessarily touching the task's status log, so a consumer +# that bounds repeated work per call cannot use the task id alone. +# bin/fm-teardown.sh asks it before its automatic # backlog close and, on 0, returns the row to Queued with its deliverable # recorded instead (bin/fm-backlog-transition-lib.sh owns that transition), so # holding the very work item a question gates is safe; only `answer` with the # captain's words or evidence-backed `reconcile close` closes the call. +# bin/fm-watch.sh asks it when an ordinary +# crew task reaches a due stale alarm - its open backlog hold need not appear in +# the task's last status line - and on a 0 bounds repeated alarms from new pane +# hashes for the decision. # # `diverged` is the read-only guard over the seam between the two records of # one captain call. See "record divergence" beside command_diverged below. @@ -1753,13 +1764,20 @@ EOF # A row this home does not carry is 3 when the caller requests the distinction; # every other read failure is a 2, printed to stderr, because a mechanical # closer must never read "cannot tell" as permission to close. -command_open() { # [--distinguish-absent] - local id=${1:-} data state distinguish_absent=0 - [ "$#" -ge 1 ] && [ "$#" -le 2 ] || { usage >&2; exit 2; } - if [ "$#" -eq 2 ]; then - [ "$2" = --distinguish-absent ] || { usage >&2; exit 2; } - distinguish_absent=1 - fi +command_open() { # [--identity] [--distinguish-absent] + local id='' identity=0 distinguish_absent=0 data state show shown_body + while [ "$#" -gt 0 ]; do + case "$1" in + --identity) identity=1 ;; + --distinguish-absent) distinguish_absent=1 ;; + -*) usage >&2; exit 2 ;; + *) + [ -z "$id" ] || { usage >&2; exit 2; } + id=$1 + ;; + esac + shift + done case "$id" in ''|*[!A-Za-z0-9._-]*) printf 'fm-captain-hold: task id must be a non-empty privacy-safe slug: %s\n' "$id" >&2 @@ -1772,6 +1790,16 @@ command_open() { # [--distinguish-absent] if fm_backlog_row_probe "$data" "$id"; then state=${FM_BACKLOG_ROW_STATE%% *} if [ "$state" != "done" ] && [ "$FM_BACKLOG_ROW_HOLD_KIND" = captain ]; then + if [ "$identity" -eq 1 ]; then + show=$(task_show "$id") || { + printf 'fm-captain-hold: captain call %s is open but its record could not be read\n' "$id" >&2 + exit 2 + } + shown_body=$(show_field "$show" body) + printf '%s#%s\n' \ + "$(body_hold_set_timestamp "$(decode_shown_value "$shown_body")")" \ + "$(resolution_record_count "$shown_body")" + fi return 0 fi return 1 diff --git a/bin/fm-classify-lib.sh b/bin/fm-classify-lib.sh index 8bd4fe746ad..fdaf0b23474 100755 --- a/bin/fm-classify-lib.sh +++ b/bin/fm-classify-lib.sh @@ -652,9 +652,9 @@ _fm_open_decisions_file_ident() { # -> strongest available identity return fi if [ "$(uname -s 2>/dev/null)" = Darwin ]; then - ident=$(LC_ALL=C stat -f '%d:%i' "$f" 2>/dev/null) || return 1 - epoch=$(LC_ALL=C stat -f '%B' "$f" 2>/dev/null) || epoch=0 - if [ "$epoch" != 0 ]; then birth=$(LC_ALL=C stat -f '%FB' "$f" 2>/dev/null) || birth=''; else birth=''; fi + ident=$(LC_ALL=C /usr/bin/stat -f '%d:%i' "$f" 2>/dev/null) || return 1 + epoch=$(LC_ALL=C /usr/bin/stat -f '%B' "$f" 2>/dev/null) || epoch=0 + if [ "$epoch" != 0 ]; then birth=$(LC_ALL=C /usr/bin/stat -f '%FB' "$f" 2>/dev/null) || birth=''; else birth=''; fi else ident=$(LC_ALL=C stat -c '%d:%i' "$f" 2>/dev/null) || return 1 epoch=$(LC_ALL=C stat -c '%W' "$f" 2>/dev/null) || epoch=0 @@ -671,7 +671,7 @@ _fm_status_file_size() { # return fi if [ "$(uname -s 2>/dev/null)" = Darwin ]; then - LC_ALL=C stat -f '%z' "$f" 2>/dev/null + LC_ALL=C /usr/bin/stat -f '%z' "$f" 2>/dev/null else LC_ALL=C stat -c '%s' "$f" 2>/dev/null fi @@ -680,7 +680,7 @@ _fm_status_file_size() { # _fm_status_file_mtime() { # local f=$1 if [ "$(uname -s 2>/dev/null)" = Darwin ]; then - LC_ALL=C stat -f '%m' "$f" 2>/dev/null + LC_ALL=C /usr/bin/stat -f '%m' "$f" 2>/dev/null else LC_ALL=C stat -c '%Y' "$f" 2>/dev/null fi @@ -1071,7 +1071,7 @@ status_presentation_marker_parse() { _status_observed_path_state() { if [ "$(uname -s 2>/dev/null)" = Darwin ]; then - LC_ALL=C stat -f '%HT:%p' "$1" 2>/dev/null + LC_ALL=C /usr/bin/stat -f '%HT:%p' "$1" 2>/dev/null else LC_ALL=C stat -c '%F:%f' "$1" 2>/dev/null fi diff --git a/bin/fm-claude-stop-autoarm.sh b/bin/fm-claude-stop-autoarm.sh index 4f24230f7fd..5f8b2d14a75 100755 --- a/bin/fm-claude-stop-autoarm.sh +++ b/bin/fm-claude-stop-autoarm.sh @@ -73,7 +73,6 @@ FM_ROOT="${FM_ROOT_OVERRIDE:-$(cd "$SCRIPT_DIR/.." && pwd)}" FM_HOME="${FM_HOME:-${FM_ROOT_OVERRIDE:-$FM_ROOT}}" STATE="${FM_STATE_OVERRIDE:-$FM_HOME/state}" CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" -GRACE=${FM_GUARD_GRACE:-300} OWNER_LOCK="$STATE/.claude-autoarm.lock" FAILURE_NOTICE="$STATE/.claude-autoarm-failure-notified" FAILURE_ALARM="$STATE/.claude-autoarm-failure-alarmed" @@ -94,6 +93,13 @@ esac # shellcheck source=bin/fm-hook-host-lib.sh . "$SCRIPT_DIR/fm-hook-host-lib.sh" +# fm-watch.sh touches the liveness beacon once per cycle, immediately before +# its terminal wait, so a healthy watcher's beacon can legitimately age up to +# FM_POLL seconds between touches (docs/turnend-guard.md "Guard grace and the +# poll cadence"). fm_poll_derived_grace (bin/fm-wake-lib.sh) is the single +# owner of that max(300, poll+60) derivation. +GRACE=${FM_GUARD_GRACE:-$(fm_poll_derived_grace)} + # Consume the Stop payload once. The decisions below are state-based; the # payload is read so a slow writer can never wedge on a full pipe, and its host # is inspected before anything else runs. @@ -217,9 +223,9 @@ while [ "$attempt" -lt "$AUTOARM_ATTEMPTS" ]; do attempt=$((attempt + 1)) OUT=$(mktemp "$STATE/.claude-autoarm-output.XXXXXX") || OUT= if [ -n "$OUT" ]; then - "$SCRIPT_DIR/fm-watch-arm.sh" >"$OUT" 2>&1 || true + FM_GUARD_GRACE="$GRACE" "$SCRIPT_DIR/fm-watch-arm.sh" >"$OUT" 2>&1 || true else - "$SCRIPT_DIR/fm-watch-arm.sh" >/dev/null 2>&1 || true + FM_GUARD_GRACE="$GRACE" "$SCRIPT_DIR/fm-watch-arm.sh" >/dev/null 2>&1 || true fi # AFK may have appeared mid-cycle: the daemon owns triage now, so suppress diff --git a/bin/fm-config-inherit-lib.sh b/bin/fm-config-inherit-lib.sh index 92d9a327fa8..de54245ad22 100644 --- a/bin/fm-config-inherit-lib.sh +++ b/bin/fm-config-inherit-lib.sh @@ -103,7 +103,7 @@ fm_config_source_present() { fm_inherit_file_mode() { if [ "$(uname)" = Darwin ]; then - stat -f %Lp "$1" 2>/dev/null + /usr/bin/stat -f %Lp "$1" 2>/dev/null else stat -c %a "$1" 2>/dev/null fi @@ -111,7 +111,7 @@ fm_inherit_file_mode() { fm_inherit_file_device() { if [ "$(uname)" = Darwin ]; then - stat -f %d "$1" 2>/dev/null + /usr/bin/stat -f %d "$1" 2>/dev/null else stat -c %d "$1" 2>/dev/null fi @@ -119,7 +119,7 @@ fm_inherit_file_device() { fm_inherit_file_link_count() { if [ "$(uname)" = Darwin ]; then - stat -f %l "$1" 2>/dev/null + /usr/bin/stat -f %l "$1" 2>/dev/null else stat -c %h "$1" 2>/dev/null fi diff --git a/bin/fm-fleet-snapshot.sh b/bin/fm-fleet-snapshot.sh index 5ef8ffc35d1..f127f7d5211 100755 --- a/bin/fm-fleet-snapshot.sh +++ b/bin/fm-fleet-snapshot.sh @@ -1086,8 +1086,8 @@ case "$FM_SNAPSHOT_SECONDMATE_LANDED_PER_HOME" in ''|*[!0-9]*) FM_SNAPSHOT_SECON # pollute arithmetic input before failing. Select the platform syntax once. if [ "$(uname 2>/dev/null || true)" = Darwin ]; then SNAPSHOT_STAT_STYLE=bsd - file_mtime_epoch() { stat -f '%m' "$1" 2>/dev/null || true; } - file_mode_octal() { stat -f '%Lp' "$1" 2>/dev/null || true; } + file_mtime_epoch() { /usr/bin/stat -f '%m' "$1" 2>/dev/null || true; } + file_mode_octal() { /usr/bin/stat -f '%Lp' "$1" 2>/dev/null || true; } else SNAPSHOT_STAT_STYLE=gnu file_mtime_epoch() { stat -c '%Y' "$1" 2>/dev/null || true; } @@ -1440,7 +1440,7 @@ bounded_parent_activities_json() { # stat_style=$6 . "$classify" if [ "$stat_style" = bsd ]; then - size=$(stat -f "%z" "$f" 2>/dev/null) || exit 3 + size=$(/usr/bin/stat -f "%z" "$f" 2>/dev/null) || exit 3 else size=$(stat -c "%s" "$f" 2>/dev/null) || exit 3 fi diff --git a/bin/fm-guard.sh b/bin/fm-guard.sh index 1e7c7202451..9dd9b20a31b 100755 --- a/bin/fm-guard.sh +++ b/bin/fm-guard.sh @@ -27,7 +27,13 @@ # bounded). Independent alarms (queued wakes, worktree tangle) are never # suppressed by that dedup. Normal wake handling (watcher briefly down between a # wake and the next supervision resume) stays inside the grace window and stays -# silent. The queued-wakes warning stays silent for the supervision branch +# silent. The queued-wakes warning counts only the rows the calling actor can +# itself present or retire (fm_wake_actor_pending_count), so it is never an +# instruction to run a drain with nothing to present. A row reserved by a live +# supervision-branch grant is never a drain instruction for main; instead of +# going silent about a visibly non-empty queue, main gets a distinct advisory +# naming the branch as the holder and saying not to drain those rows. +# The ordinary warning also stays silent for the supervision branch # actor (FM_SUPERVISION_ACTOR=branch), because that actor runs guarded commands # while handling exactly the queued rows its grant covers and can drain nothing # else. Always exits 0: the guard warns, it never blocks. @@ -41,6 +47,7 @@ CONFIG="${FM_CONFIG_OVERRIDE:-$FM_HOME/config}" WATCH="$SCRIPT_DIR/fm-watch.sh" GRACE=${FM_GUARD_GRACE:-300} queue_pending=false +queue_branch_held=false READ_ONLY=${FM_GUARD_READ_ONLY:-0} case "$READ_ONLY" in 1|true|TRUE|yes|YES) READ_ONLY=1 ;; *) READ_ONLY=0 ;; esac CONTINUE_LINE=${FM_GUARD_CONTINUE_LINE:-This is a supervision warning only; the guarded operation WILL still run.} @@ -177,7 +184,18 @@ if [ "$needed" = false ]; then exit 0 fi -[ -s "$FM_WAKE_QUEUE" ] && queue_pending=true +# Count only the rows this actor could actually present or retire, so the +# warning never sends an actor to a drain that provably has nothing for it. +# fm-wake-lib.sh owns that per-actor classification. A non-empty queue with +# nothing for main is the branch-held case: keep the raw pending signal visible +# there as its own advisory rather than dropping it. +if [ -s "$FM_WAKE_QUEUE" ]; then + if [ "$(fm_wake_actor_pending_count "$GUARD_ACTOR")" -gt 0 ]; then + queue_pending=true + elif [ "$GUARD_ACTOR" != branch ] && [ "$(fm_wake_actor_pending_count branch)" -gt 0 ]; then + queue_branch_held=true + fi +fi # No fresh watcher with tasks in flight is the dangerous state: emit a prominent, # bordered banner FIRST so it reads as an alarm, not a buried stderr line. Later @@ -256,5 +274,7 @@ if "$queue_pending"; then elif [ "$GUARD_ACTOR" != branch ]; then echo "WARNING: queued wakes pending - drain them with bin/fm-wake-drain.sh before anything else." >&2 fi +elif "$queue_branch_held"; then + echo "NOTICE: wake rows held by the live supervision branch - it presents and acknowledges them; do not drain them from here." >&2 fi exit 0 diff --git a/bin/fm-inactive-reconcile.sh b/bin/fm-inactive-reconcile.sh index 52064545836..9c30a9074be 100755 --- a/bin/fm-inactive-reconcile.sh +++ b/bin/fm-inactive-reconcile.sh @@ -121,7 +121,7 @@ if [ "$FM_INACTIVE_RECONCILE_BUDGET_SECS" -gt 30 ]; then fi if [ "$(uname)" = Darwin ]; then - file_mtime() { stat -f %m "$1" 2>/dev/null; } + file_mtime() { /usr/bin/stat -f %m "$1" 2>/dev/null; } else file_mtime() { stat -c %Y "$1" 2>/dev/null; } fi diff --git a/bin/fm-lock-lib.sh b/bin/fm-lock-lib.sh index f3b070ec8cf..7303ac571ad 100644 --- a/bin/fm-lock-lib.sh +++ b/bin/fm-lock-lib.sh @@ -24,7 +24,7 @@ fm_lock_log() { # no wake-queue machinery when a caller only needs the staleness proof. fm_lock_path_mtime() { if [ "$(uname)" = Darwin ]; then - stat -f %m "$1" 2>/dev/null + /usr/bin/stat -f %m "$1" 2>/dev/null else stat -c %Y "$1" 2>/dev/null fi diff --git a/bin/fm-pending-reply-lib.sh b/bin/fm-pending-reply-lib.sh index 7e4b00e4948..27c9bc9e90d 100755 --- a/bin/fm-pending-reply-lib.sh +++ b/bin/fm-pending-reply-lib.sh @@ -571,7 +571,7 @@ fm_pending_reply_file_signature() { # local path=$1 [ -f "$path" ] || { printf 'missing'; return 0; } if [ "$(uname -s 2>/dev/null)" = Darwin ]; then - LC_ALL=C stat -f '%d:%i:%z:%m:%c' "$path" 2>/dev/null || printf 'unreadable' + LC_ALL=C /usr/bin/stat -f '%d:%i:%z:%m:%c' "$path" 2>/dev/null || printf 'unreadable' else LC_ALL=C stat -c '%d:%i:%s:%Y:%Z' "$path" 2>/dev/null || printf 'unreadable' fi diff --git a/bin/fm-pr-lib.sh b/bin/fm-pr-lib.sh index d9580dc9b4a..610def7599d 100755 --- a/bin/fm-pr-lib.sh +++ b/bin/fm-pr-lib.sh @@ -215,7 +215,7 @@ fm_pr_head_valid() { fm_pr_file_mode() { if [ "$(uname)" = Darwin ]; then - stat -f %Lp "$1" 2>/dev/null + /usr/bin/stat -f %Lp "$1" 2>/dev/null else stat -c %a "$1" 2>/dev/null fi @@ -223,7 +223,7 @@ fm_pr_file_mode() { fm_pr_file_device() { if [ "$(uname)" = Darwin ]; then - stat -f %d "$1" 2>/dev/null + /usr/bin/stat -f %d "$1" 2>/dev/null else stat -c %d "$1" 2>/dev/null fi @@ -231,7 +231,7 @@ fm_pr_file_device() { fm_pr_file_link_count() { if [ "$(uname)" = Darwin ]; then - stat -f %l "$1" 2>/dev/null + /usr/bin/stat -f %l "$1" 2>/dev/null else stat -c %h "$1" 2>/dev/null fi @@ -239,7 +239,7 @@ fm_pr_file_link_count() { fm_pr_file_inode() { if [ "$(uname)" = Darwin ]; then - stat -f %i "$1" 2>/dev/null + /usr/bin/stat -f %i "$1" 2>/dev/null else stat -c %i "$1" 2>/dev/null fi diff --git a/bin/fm-procevent-extension-capture.pl b/bin/fm-procevent-extension-capture.pl index 3e877ae6c43..485c1a0343d 100644 --- a/bin/fm-procevent-extension-capture.pl +++ b/bin/fm-procevent-extension-capture.pl @@ -1,7 +1,7 @@ use strict; use warnings; use Cwd qw(getcwd); -use Fcntl qw(O_CREAT O_EXCL O_NOFOLLOW O_RDONLY O_RDWR); +use Fcntl qw(O_CREAT O_EXCL O_NOFOLLOW O_RDONLY O_RDWR O_WRONLY); use JSON::PP qw(encode_json); use POSIX qw(dup2); @@ -92,8 +92,12 @@ my ($registry_fd, $inbox_fd, $reservation_fd, $id, $adapter, $extension_id, $extension_version, $capability_version, $package_digest, $binding_digest, $claim_token, $runner_name, $output_name, $runner_pid, $claim_identity, $limit, @command) = @ARGV; +my $launch_ready_name; +$launch_ready_name = shift @command if @command && $command[0] ne "--"; die "missing command\n" unless @command && shift(@command) eq "--"; die "invalid limit\n" unless defined $limit && $limit =~ /\A\d+\z/; +die "invalid launch boundary\n" if defined($launch_ready_name) + && $launch_ready_name !~ /\A\.[A-Za-z0-9._-]{1,384}\.launch-ready\z/; our ($registry_dir, $registry, $reservation_dir, $reservation_root, $sequence); sub fail { die "capture failed: $_[0]\n"; } @@ -180,6 +184,14 @@ sub write_reservation { write_all($runner, "$runner_pid\n"); close($runner) or fail("cannot close runner record"); my $stage = open_new($output_name); +my $launch_ready; +if (defined $launch_ready_name) { + sysopen($launch_ready, $launch_ready_name, O_WRONLY | O_NOFOLLOW) + or fail("cannot open launch boundary"); + my @launch_ready_stat = stat($launch_ready); + fail("unsafe launch boundary") unless @launch_ready_stat && -f _ && $launch_ready_stat[4] == $< + && ($launch_ready_stat[2] & 07777) == 0600 && $launch_ready_stat[3] == 1; +} pipe(my $reader, my $writer) or fail("cannot create output pipe"); my $child = fork(); defined $child or fail("cannot fork adapter"); @@ -191,6 +203,10 @@ sub write_reservation { exit 127; } close($writer); +if (defined $launch_ready) { + write_all($launch_ready, "ready\n"); + close($launch_ready) or fail("cannot close launch boundary"); +} my ($written, $truncated) = (0, 0); while (1) { my $read = sysread($reader, my $buffer, 65536); diff --git a/bin/fm-procevent-lavish.sh b/bin/fm-procevent-lavish.sh index 3db763d13de..ece166e7d37 100755 --- a/bin/fm-procevent-lavish.sh +++ b/bin/fm-procevent-lavish.sh @@ -97,7 +97,7 @@ # That is an internal retry, not news, so registering the raw poll made the # generic runner capture it and wake the whole fleet. `poll` therefore re-runs # the published poll up to POLL_RETRY_LIMIT times for that exact response, with -# POLL_RETRY_DELAY_DEFAULT seconds between attempts. The match is exact and +# attempt starts at least POLL_RETRY_DELAY_DEFAULT seconds apart. The match is exact and # deliberately narrow: real feedback, ended and missing sessions, any other # SERVER_ERROR, and the same interruption still standing after the bound is # spent are all printed straight through and captured normally. The retry is a @@ -175,6 +175,7 @@ cmd_retire() { # without waiting it out. POLL_RETRY_LIMIT=12 POLL_RETRY_DELAY_DEFAULT=5 +POLL_RETRY_DELAY_MIN=1 POLL_RETRY_DELAY_MAX=60 # Exit 0 only for the exact two-line interruption, and nothing else. The whole @@ -227,8 +228,8 @@ poll_response_filter() { # ' "$1" } -# Seconds between retries. FM_LAVISH_POLL_RETRY_DELAY is a bounded test -# override; a malformed or out-of-range value is refused rather than quietly +# Minimum seconds between retry attempt starts. FM_LAVISH_POLL_RETRY_DELAY is a +# bounded test override; a malformed or out-of-range value is refused rather than quietly # rounded, because silently changing a retry cadence is how a bound stops # meaning anything. poll_retry_delay() { @@ -238,15 +239,28 @@ poll_retry_delay() { return 0 fi case "$delay" in - *[!0-9]*) die "FM_LAVISH_POLL_RETRY_DELAY must be whole seconds from 0 to $POLL_RETRY_DELAY_MAX: $delay" ;; + *[!0-9]*) die "FM_LAVISH_POLL_RETRY_DELAY must be whole seconds from $POLL_RETRY_DELAY_MIN to $POLL_RETRY_DELAY_MAX: $delay" ;; esac - [ "$delay" -le "$POLL_RETRY_DELAY_MAX" ] \ - || die "FM_LAVISH_POLL_RETRY_DELAY must be whole seconds from 0 to $POLL_RETRY_DELAY_MAX: $delay" + [ "$delay" -ge "$POLL_RETRY_DELAY_MIN" ] && [ "$delay" -le "$POLL_RETRY_DELAY_MAX" ] \ + || die "FM_LAVISH_POLL_RETRY_DELAY must be whole seconds from $POLL_RETRY_DELAY_MIN to $POLL_RETRY_DELAY_MAX: $delay" printf '%s\n' "$delay" } +poll_iteration_started() { + perl -MTime::HiRes=clock_gettime,CLOCK_MONOTONIC -e \ + 'printf "%.6f\\n", clock_gettime(CLOCK_MONOTONIC)' +} + +poll_iteration_floor_wait() { + perl -MTime::HiRes=clock_gettime,sleep,CLOCK_MONOTONIC -e ' + my ($started, $floor) = @ARGV; + my $remaining = $floor - (clock_gettime(CLOCK_MONOTONIC) - $started); + sleep($remaining) if $remaining > 0; + ' "$1" "$2" +} + cmd_poll() { - local artifact=${1-} delay attempt=0 response cleanup_command rc filter_rc + local artifact=${1-} delay attempt=0 response cleanup_command rc filter_rc iteration_started local pipeline_status [ -n "$artifact" ] || usage [ "$#" -eq 1 ] || usage @@ -266,6 +280,7 @@ cmd_poll() { trap "$cleanup_command; trap - $signal; kill -$signal $$" "$signal" done while :; do + iteration_started=$(poll_iteration_started) || die "cannot start the poll rate governor" lavish-axi poll "$artifact" | poll_response_filter "$response" pipeline_status=("${PIPESTATUS[@]}") rc=${pipeline_status[0]} @@ -275,7 +290,8 @@ cmd_poll() { 10) if [ "$attempt" -lt "$POLL_RETRY_LIMIT" ]; then attempt=$((attempt + 1)) - sleep "$delay" + poll_iteration_floor_wait "$iteration_started" "$delay" \ + || die "cannot enforce the poll rate governor" else cat -- "$response" break diff --git a/bin/fm-procevent-lib.sh b/bin/fm-procevent-lib.sh index 01a48df901c..ff8ef62a920 100644 --- a/bin/fm-procevent-lib.sh +++ b/bin/fm-procevent-lib.sh @@ -103,6 +103,202 @@ fm_procevent_any_registered() { return 1 } +# --- owning-session lease --------------------------------------------------- +# A runner is detached into its own process group so it survives the turn that +# started it. That is what makes a persistent source work, and on its own it is +# also what lets a runner outlive its whole home: once reparented to init, +# nothing bounds its lifetime, so its blocking child - and everything that child +# spawns - can keep running indefinitely. +# +# The bound is a lease on the OWNING STATE ROOT. Owner-presence operations +# refresh it, an attached public start keeps it fresh while its caller remains +# attached, and the watcher's reconcile cycle keeps it fresh in a live home. +# A guard proves the runner's owner is still there by reading that lease from +# the physical state root recorded in the claim. After two consecutive checks +# cannot prove both the root identity and a fresh lease, it stops the runner's +# process group. The lease is keyed by state root, so another home's live runner +# is untouched: that home refreshes its own lease. Nothing here keys on a script +# name, a command line, or a process name, all of which are shared across homes. + +fm_procevent_owner_lease_path() { # + printf '%s/.owner-lease\n' "$(fm_procevent_registry_dir "$1")" +} + +# Record owner-presence activity in this home's process-event state. Best +# effort by design: a home with no registry directory yet owns no runner. +fm_procevent_owner_lease_touch() { # + local reg lease tmp now + reg=$(fm_procevent_registry_dir "$1") + [ -d "$reg" ] && [ ! -L "$reg" ] || return 1 + lease=$(fm_procevent_owner_lease_path "$1") + now=$(perl -MTime::HiRes=clock_gettime,CLOCK_MONOTONIC -e \ + 'printf "%.6f\n", clock_gettime(CLOCK_MONOTONIC)') || return 1 + tmp=$(umask 077; mktemp "$reg/.owner-lease.XXXXXX") || return 1 + if ! printf '%s\n' "$now" > "$tmp" || ! mv -f -- "$tmp" "$lease"; then + rm -f -- "$tmp" + return 1 + fi +} + +# Seconds since the last refresh. Fails when the lease is absent or unreadable, +# which is what a removed home looks like from inside a surviving runner. +fm_procevent_owner_lease_age() { # + local lease value + lease=$(fm_procevent_owner_lease_path "$1") + [ -f "$lease" ] && [ ! -L "$lease" ] || return 1 + IFS= read -r value < "$lease" || return 1 + perl -MTime::HiRes=clock_gettime,CLOCK_MONOTONIC -e ' + use strict; + use warnings; + my $value = shift; + $value =~ /\A[0-9]+(?:\.[0-9]+)?\z/ or exit 1; + my $now = clock_gettime(CLOCK_MONOTONIC); + $now >= $value or exit 1; + printf "%d\n", int($now - $value); + ' "$value" +} + +# How long a runner keeps going with no activity in its owning home. The default +# is forty watcher cycles at the default poll interval, so an ordinary busy or +# briefly wedged home never trips it, while a home that is simply gone stops +# owning processes within the hour rather than within a day. +FM_PROCEVENT_OWNER_LEASE_DEFAULT_SECONDS=600 +FM_PROCEVENT_OWNER_LEASE_MIN_SECONDS=1 +FM_PROCEVENT_OWNER_LEASE_MAX_SECONDS=86400 + +fm_procevent_owner_lease_seconds() { + local value=${FM_PROCEVENT_OWNER_LEASE_SECONDS-} + if [ -z "$value" ]; then + printf '%s\n' "$FM_PROCEVENT_OWNER_LEASE_DEFAULT_SECONDS" + return 0 + fi + case "$value" in ''|*[!0-9]*) return 1 ;; esac + [ "$value" -ge "$FM_PROCEVENT_OWNER_LEASE_MIN_SECONDS" ] || return 1 + [ "$value" -le "$FM_PROCEVENT_OWNER_LEASE_MAX_SECONDS" ] || return 1 + printf '%s\n' "$value" +} + +# How often a runner's guard re-reads that lease. One watcher cycle at the +# default poll interval, so the guard costs about as much as the cycle that +# refreshes what it reads. +FM_PROCEVENT_OWNER_CHECK_DEFAULT_SECONDS=15 +FM_PROCEVENT_OWNER_CHECK_MIN_SECONDS=1 +FM_PROCEVENT_OWNER_CHECK_MAX_SECONDS=3600 + +fm_procevent_owner_check_seconds() { + local value=${FM_PROCEVENT_OWNER_CHECK_SECONDS-} + if [ -z "$value" ]; then + printf '%s\n' "$FM_PROCEVENT_OWNER_CHECK_DEFAULT_SECONDS" + return 0 + fi + case "$value" in ''|*[!0-9]*) return 1 ;; esac + [ "$value" -ge "$FM_PROCEVENT_OWNER_CHECK_MIN_SECONDS" ] || return 1 + [ "$value" -le "$FM_PROCEVENT_OWNER_CHECK_MAX_SECONDS" ] || return 1 + printf '%s\n' "$value" +} + +FM_PROCEVENT_LAUNCH_FLOOR_DEFAULT_SECONDS=1 +FM_PROCEVENT_LAUNCH_FLOOR_MIN_SECONDS=1 +FM_PROCEVENT_LAUNCH_FLOOR_MAX_SECONDS=3600 + +fm_procevent_launch_floor_seconds() { + local value=${FM_PROCEVENT_LAUNCH_FLOOR_SECONDS-} + if [ -z "$value" ]; then + printf '%s\n' "$FM_PROCEVENT_LAUNCH_FLOOR_DEFAULT_SECONDS" + return 0 + fi + case "$value" in ''|*[!0-9]*) return 1 ;; esac + [ "$value" -ge "$FM_PROCEVENT_LAUNCH_FLOOR_MIN_SECONDS" ] || return 1 + [ "$value" -le "$FM_PROCEVENT_LAUNCH_FLOOR_MAX_SECONDS" ] || return 1 + printf '%s\n' "$value" +} + +fm_procevent_launch_floor_reset_locked() { # + local reg identity + case "$3" in *:*) ;; *) return 1 ;; esac + case "$3" in ''|*[!0-9:]*) return 1 ;; esac + reg=$(fm_procevent_registry_dir "$1") || return 1 + identity=${3//:/-} + rm -f -- "$reg/$2.$identity.last-launch" +} + +fm_procevent_launch_floor_prune_locked() { # + local reg identity keep stamp + case "$3" in *:*) ;; *) return 1 ;; esac + case "$3" in ''|*[!0-9:]*) return 1 ;; esac + reg=$(fm_procevent_registry_dir "$1") || return 1 + identity=${3//:/-} + keep="$reg/$2.$identity.last-launch" + for stamp in "$reg/$2".*.last-launch "$reg/$2.last-launch"; do + [ "$stamp" = "$keep" ] && continue + [ -e "$stamp" ] || [ -L "$stamp" ] || continue + rm -f -- "$stamp" || return 1 + done +} + +fm_procevent_launch_floor_wait() { # + local state=$1 id=$2 expected=$3 floor=$4 reg stamp identity registration current_identity status=0 + case "$expected" in *:*) ;; *) return 1 ;; esac + case "$expected" in ''|*[!0-9:]*) return 1 ;; esac + reg=$(fm_procevent_registry_dir "$state") || return 1 + identity=${expected//:/-} + stamp="$reg/$id.$identity.last-launch" + [ ! -L "$stamp" ] || return 1 + [ ! -e "$stamp" ] || [ -f "$stamp" ] || return 1 + perl -MTime::HiRes=clock_gettime,sleep,CLOCK_MONOTONIC -e ' + use strict; + use warnings; + my ($path, $floor) = @ARGV; + my $previous; + if (-e $path) { + open my $in, "<", $path or exit 1; + my $value = <$in>; + close $in or exit 1; + defined($value) && $value =~ /\A([0-9]+(?:\.[0-9]+)?)\n?\z/ or exit 1; + $previous = 0 + $1; + } + my $now = clock_gettime(CLOCK_MONOTONIC); + my $elapsed = defined($previous) && $now >= $previous ? $now - $previous : undef; + sleep($floor - $elapsed) if defined($elapsed) && $elapsed < $floor; + ' "$stamp" "$floor" || return 1 + + # Registration publication holds this same source lock while replacing and + # pruning pacing state, so a superseded sleeper cannot recreate its stamp. + fm_procevent_source_lock_acquire "$id" || return 1 + registration="$reg/$id.source" + current_identity=$(fm_pr_file_identity "$registration" 2>/dev/null) || current_identity= + if [ "$current_identity" != "$expected" ]; then + fm_procevent_source_lock_release "$id" || return 1 + return 2 + fi + [ ! -L "$stamp" ] && { [ ! -e "$stamp" ] || [ -f "$stamp" ]; } || status=1 + if [ "$status" -eq 0 ]; then + perl -MTime::HiRes=clock_gettime,CLOCK_MONOTONIC -MFcntl=:DEFAULT -e ' + use strict; + use warnings; + my $path = shift; + my $now = clock_gettime(CLOCK_MONOTONIC); + my $tmp = "$path.$$"; + sysopen(my $out, $tmp, O_WRONLY | O_CREAT | O_EXCL, 0600) or exit 1; + print {$out} "$now\n" or exit 1; + close $out or exit 1; + rename $tmp, $path or exit 1; + ' "$stamp" || status=1 + fi + if [ "$status" -ne 0 ]; then + fm_procevent_source_lock_release "$id" || : + return "$status" + fi + return 0 +} + +# True while the owning home is provably still active. +fm_procevent_owner_alive() { # + local age + age=$(fm_procevent_owner_lease_age "$1") || return 1 + [ "$age" -le "$2" ] +} + # --- ownership -------------------------------------------------------------- # A claim is a private file recording the home, runner pid, claim generation, # and process identity. Registration and every ownership transition are @@ -130,7 +326,7 @@ fm_procevent_source_lock_release() { } fm_procevent_registration_publish_locked() { # - local state=$1 adapter=$2 id=$3 reg dest tmp arg + local state=$1 adapter=$2 id=$3 reg dest tmp arg identity shift 3 fm_procevent_adapter_valid "$adapter" || return 1 fm_procevent_source_id_valid "$id" || return 1 @@ -148,7 +344,11 @@ fm_procevent_registration_publish_locked() { # "$tmp" && chmod 0600 "$tmp" && mv -f -- "$tmp" "$dest"; then + } > "$tmp" && chmod 0600 "$tmp" \ + && identity=$(fm_pr_file_identity "$tmp") \ + && fm_procevent_launch_floor_reset_locked "$state" "$id" "$identity" \ + && mv -f -- "$tmp" "$dest"; then + fm_procevent_launch_floor_prune_locked "$state" "$id" "$identity" 2>/dev/null || : return 0 fi rm -f -- "$tmp" @@ -160,7 +360,7 @@ fm_procevent_registration_publish_locked() { # local state=$1 adapter=$2 id=$3 extension_id=$4 extension_version=$5 capability_version=$6 - local package_digest=$7 binding_digest=$8 config_ref=$9 registration_token=${10} reg dest tmp + local package_digest=$7 binding_digest=$8 config_ref=$9 registration_token=${10} reg dest tmp identity fm_procevent_adapter_valid "$adapter" || return 1 fm_procevent_source_id_valid "$id" || return 1 fm_procevent_extension_id_valid "$extension_id" || return 1 @@ -188,7 +388,11 @@ fm_procevent_extension_registration_publish_locked() { # "$tmp" && chmod 0600 "$tmp" && mv -f -- "$tmp" "$dest"; then + } > "$tmp" && chmod 0600 "$tmp" \ + && identity=$(fm_pr_file_identity "$tmp") \ + && fm_procevent_launch_floor_reset_locked "$state" "$id" "$identity" \ + && mv -f -- "$tmp" "$dest"; then + fm_procevent_launch_floor_prune_locked "$state" "$id" "$identity" 2>/dev/null || : return 0 fi rm -f -- "$tmp" @@ -385,7 +589,7 @@ fm_procevent_claim_capture_reservation_remove_locked() { # independently has no members left. The separate group check also covers a # reused live pid whose identity differs while the old generation survives. # A live matched owner (state 0), an unreadable identity (state 2), and a -# crashed leader whose owned group is still running (state 3) all return false. +# crashed leader with a still-live ambiguous group (state 3) all return false. fm_procevent_claim_generation_gone_locked() { local state=0 fm_procevent_pid_state "${FM_PROCEVENT_CLAIM_PID:-}" "${FM_PROCEVENT_CLAIM_IDENTITY:-}" || state=$? @@ -408,26 +612,21 @@ fm_procevent_claim_capture_reservation_reclaim_locked() { } # fm_procevent_group_alive -# True while any process remains in the process group a runner leads. A runner -# started by reconcile is its own group leader, so this is what distinguishes a -# generation that is really gone from one whose leader died while its blocking -# source child kept running. +# True while any process remains in the runner's numeric process group. A runner +# starts as its own group leader, but after that leader exits a same-numbered +# group may be reused, so group presence prevents proving the generation gone. fm_procevent_group_alive() { case "$1" in ''|*[!0-9]*) return 1 ;; esac kill -0 -"$1" 2>/dev/null } # fm_procevent_pid_state -# 0 live match, 1 stale, 2 uncertain, 3 orphaned group. +# 0 live match, 1 stale, 2 uncertain, 3 ambiguous leaderless group. # -# State 3 is the crash cut: the runner leader is gone, but its owned process -# group still has members, so the old generation can still be consuming the -# source. Treating that as stale would release ownership and let a second -# poller start against one canonical source. Only the leader being absent -# reaches state 3, which is also what makes signalling that group safe: if this -# pid had been reused by an unrelated process the leader would be alive, so the -# identity comparison below would classify it stale or uncertain and no group -# signal would ever follow. +# State 3 is the crash cut: the runner leader is gone, but a process group with +# its numeric id still has members. That group may be the old generation or a +# leaderless group created after PID/PGID reuse, so cleanup preserves the claim +# without signalling the group or starting a replacement. fm_procevent_pid_state() { local pid=$1 expected=$2 actual if ! fm_pid_alive "$pid"; then @@ -526,6 +725,11 @@ fm_procevent_claim_acquire_locked() { if [ "$status" -eq 0 ]; then FM_PROCEVENT_CLAIM_TOKEN=$token FM_PROCEVENT_CLAIM_REG_IDENTITY=$reg_identity + FM_PROCEVENT_CLAIM_STATE_ROOT=$state_root + FM_PROCEVENT_CLAIM_STATE_DEVICE=$state_device + FM_PROCEVENT_CLAIM_STATE_INODE=$state_inode + FM_PROCEVENT_CLAIM_STATE_OWNER=$state_owner + FM_PROCEVENT_CLAIM_STATE_MODE=$state_mode fi fi [ "$status" -eq 0 ] || { [ -z "${tmp:-}" ] || rm -f -- "$tmp"; } @@ -642,7 +846,7 @@ fm_procevent_path_normalize() { fm_procevent_directory_owned_by_current_user() { local owner if [ "$(uname)" = Darwin ]; then - owner=$(stat -f %u "$1" 2>/dev/null) + owner=$(/usr/bin/stat -f %u "$1" 2>/dev/null) else owner=$(stat -c %u "$1" 2>/dev/null) fi diff --git a/bin/fm-procevent.sh b/bin/fm-procevent.sh index f01480ceb36..aaa7d60c541 100755 --- a/bin/fm-procevent.sh +++ b/bin/fm-procevent.sh @@ -146,13 +146,35 @@ # while ACTING on it is firstmate's judgement, so the capture stays unacknowledged # and its `check` wake reaches the handler exactly as it would have anyway. # +# A runner is bound to the HOME that owns it, not to the one session that armed +# it: a persistent source is meant to outlive that session, so reconcile stops a +# runner whose source is retired in a live home, and this lease is the backstop +# for a home that is GONE. Detaching a runner into its own +# process group is what lets a persistent source outlive the turn that armed it, +# and with nothing else it is also what lets a runner outlive its whole home: +# reparented to init, it keeps its blocking child - and everything that child +# spawns - running with nobody left to reap it. So every runner starts a small +# guard beside it, in its own separate process group, which re-reads the owning +# state root's lease on a bounded cadence and stops the runner's whole process +# group once that lease can no longer be proved fresh. Owner-presence operations +# refresh the lease, an attached public start keeps it fresh while its caller +# remains attached, and the watcher's reconcile cycle keeps it fresh in a live +# home. A runner exports the inherited FM_PROCEVENT_IN_RUNNER marker and every +# refresh is skipped under it, so a runner and its ordinary children do not +# certify their own owner. That rule is CONFUSED-AGENT-GRADE, the grade +# bin/fm-lease-lib.sh documents: a source that DELIBERATELY strips the marker +# can still refresh, and adversarial-grade unforgeability is out of scope (see +# docs/configuration.md). Scope is the owning state root and one runner +# generation, never a script or process name, so a live source in +# another home is untouched. See bin/fm-procevent-lib.sh for the lease itself. +# # Ownership is machine-wide per canonical source, because separate Firstmate # homes can share one underlying source store. A live owner is never displaced; # only a claim whose stale owner and independently absent process group prove -# its whole generation gone is reclaimed. A runner leads its own process group, -# so a crashed leader or reused pid whose old group still has members cannot -# relax ownership cleanup. Reconcile stops a safely identified surviving group -# before replacement and keeps the claim for a later retry when it cannot. +# its whole generation gone is reclaimed. A crashed leader or reused pid whose +# process group still has members cannot relax ownership cleanup. Reconcile +# signals only a live identity-matched runner group and otherwise keeps the +# claim without starting a replacement. # # Durability boundary: see bin/fm-procevent-lib.sh. This runner proves capture # before publication and bounded re-announcement until handled, and nothing @@ -437,6 +459,7 @@ cmd_register() { die "cannot publish the registration" fi fm_procevent_source_lock_release "$id" + owner_lease_refresh printf 'registered: %s (%s)\n' "$id" "$adapter" } @@ -526,6 +549,7 @@ cmd_register_extension() { fi fm_procevent_source_lock_release "$id" extension_lifecycle_lock_release + owner_lease_refresh printf 'registered: %s (%s from %s@%s)\n' "$id" "$adapter" "$extension_id" "$extension_version" printf 'owner-token: %s\n' "$registration_token" printf 'retire: bin/fm-procevent.sh retire %s --if-owner %s\n' "$id" "$registration_token" @@ -585,8 +609,15 @@ publish_pending() { # [result-file-to-skip] printf '%s\n' "$published" } -isolate_runner() { # - local mode=$1 id=$2 program +# Start one command as the leader of a fresh process group, either waiting for +# it (the public `start` boundary) or detaching from it (reconcile's restart and +# the runner's own owner guard). The guard deliberately gets its OWN group +# rather than joining the runner's: it has to survive the group signal it sends, +# and a member of the runner's group would also make that group read as alive +# after the runner itself is gone. +isolate_process() { # [argv...] + local mode=$1 program + shift # shellcheck disable=SC2016 # Perl owns every $ expression in this literal program. program='my $mode = shift @ARGV; defined(my $pid = fork) or exit 125; @@ -602,27 +633,66 @@ isolate_runner() { # exit(128 + ($status & 127)) if $status & 127; exit($status >> 8);' if [ "$mode" = wait ]; then - exec perl -e "$program" "$mode" "$SCRIPT_DIR/fm-procevent.sh" _start "$id" + perl -e "$program" "$mode" "$@" + return $? fi - perl -e "$program" "$mode" "$SCRIPT_DIR/fm-procevent.sh" _start "$id" >/dev/null 2>&1 & + perl -e "$program" "$mode" "$@" >/dev/null 2>&1 & +} + +isolate_runner() { # + isolate_process "$1" "$SCRIPT_DIR/fm-procevent.sh" _start "$2" } -require_runner_group() { - local pgid +require_isolated_group() { # + local role=$1 pgid [ "${FM_PROCEVENT_RUNNER_GROUP:-}" = "$$" ] \ - || die "runner process group was not isolated" + || die "$role process group was not isolated" pgid=$(ps -o pgid= -p "$$" 2>/dev/null | tr -d '[:space:]') \ - || die "cannot inspect runner process group" - [ -n "$pgid" ] || die "cannot inspect runner process group" - [ "$pgid" = "$$" ] || die "runner does not lead its process group" + || die "cannot inspect $role process group" + [ -n "$pgid" ] || die "cannot inspect $role process group" + [ "$pgid" = "$$" ] || die "$role does not lead its process group" unset FM_PROCEVENT_RUNNER_GROUP } +require_runner_group() { require_isolated_group runner; } + +# Record owner-presence activity for this home. Skipped under the inherited +# FM_PROCEVENT_IN_RUNNER marker, so a runner and its ordinary children do not +# keep refreshing their own lease after the home goes away. +# Confused-agent-grade: a source that deliberately unsets the marker can still +# refresh, and that is out of scope (see docs/configuration.md). +owner_lease_refresh() { + [ "${FM_PROCEVENT_IN_RUNNER:-0}" = 1 ] && return 0 + fm_procevent_owner_lease_touch "$STATE" 2>/dev/null || true +} + +owner_lease_keepalive() { # + local parent=$1 identity=$2 state + while :; do + sleep 1 + fm_procevent_pid_state "$parent" "$identity" + state=$? + case "$state" in + 0) owner_lease_refresh ;; + 2) ;; + *) return 0 ;; + esac + done +} + cmd_start_public() { - local id=${1-} + local id=${1-} identity keeper status [ "$#" -eq 1 ] || usage fm_procevent_source_id_valid "$id" || die "source id must be path-safe: $id" + owner_lease_refresh + identity=$(fm_pid_identity "$$" 2>/dev/null) || die "cannot identify the attached owner" + owner_lease_keepalive "$$" "$identity" & + keeper=$! isolate_runner wait "$id" + status=$? + kill "$keeper" 2>/dev/null || true + wait "$keeper" 2>/dev/null || true + return "$status" } cmd_start() { @@ -681,6 +751,10 @@ cmd_start() { die "extension registration owner is unreadable: $id" ;; esac + exec 7<"$(source_file "$id")" || { + fm_procevent_source_lock_release "$id" + die "cannot retain registration identity: $id" + } fm_procevent_claim_acquire_locked "$id" "$FM_HOME" "$$" "$(source_file "$id")" "$STATE" claimed=$? fm_procevent_source_lock_release "$id" @@ -694,6 +768,8 @@ cmd_start() { CLAIM_PID=$$ CLAIM_TOKEN=$FM_PROCEVENT_CLAIM_TOKEN CLAIM_REG_IDENTITY=$FM_PROCEVENT_CLAIM_REG_IDENTITY + CLAIM_STATE_DEVICE=$FM_PROCEVENT_CLAIM_STATE_DEVICE + CLAIM_STATE_INODE=$FM_PROCEVENT_CLAIM_STATE_INODE STAGED_OUTPUT= release_start_claim() { extension_lifecycle_lock_release 2>/dev/null || true @@ -711,7 +787,14 @@ cmd_start() { fm_procevent_source_lock_release "$CLAIM_ID" 2>/dev/null || true } trap release_start_claim EXIT - local runner inbox reservation_dir staging + # The inherited marker keeps the runner and its ordinary children from + # accidentally refreshing the owner lease. A source that deliberately strips + # it is outside this confused-agent-grade boundary. + export FM_PROCEVENT_IN_RUNNER=1 + start_owner_guard "$id" || die "cannot start the runner's owner guard: $id" + local launch_floor runner inbox reservation_dir staging launch_ready launch_reply launch_pid + launch_floor=$(fm_procevent_launch_floor_seconds) \ + || die "FM_PROCEVENT_LAUNCH_FLOOR_SECONDS must be whole seconds from $FM_PROCEVENT_LAUNCH_FLOOR_MIN_SECONDS to $FM_PROCEVENT_LAUNCH_FLOOR_MAX_SECONDS" if [ "$extension_owner" -eq 1 ]; then staging=$(fm_procevent_extension_staging_prepare "$STATE") \ || die "cannot safely prepare the external registry staging boundary" @@ -748,13 +831,45 @@ cmd_start() { # Built-in adapters do not run the extension capture helper, so keep this # sentinel defined while sharing the no-result branch below under `set -u`. local truncated=0 capture_state='' durable='' reservation_terminal='' reservation_silent='' + fm_procevent_launch_floor_wait "$STATE" "$id" "$CLAIM_REG_IDENTITY" "$launch_floor" + case "$?" in + 0) ;; + # A superseded generation leaves nothing behind. The runner marker is + # written before this wait, and a home sweep counts a marker with no owned + # claim as a preflight failure, so exiting without removing it would make + # that home refuse to sweep. + 2) [ "$extension_owner" -eq 1 ] || rm -f -- "$runner"; exit 0 ;; + *) die "cannot enforce the source launch floor: $id" ;; + esac + exec 7<&- if [ "$extension_owner" -eq 1 ]; then - capture_state=$(perl "$SCRIPT_DIR/fm-procevent-extension-capture.pl" \ + launch_ready=".$id.$CLAIM_TOKEN.launch-ready" + launch_reply="$REG/.$id.$CLAIM_TOKEN.launch-reply" + (umask 077; : > "$REG/$launch_ready" && : > "$launch_reply") || { + rm -f -- "$REG/$launch_ready" "$launch_reply" + fm_procevent_source_lock_release "$id" + die "cannot prepare the source launch boundary: $id" + } + perl "$SCRIPT_DIR/fm-procevent-extension-capture.pl" \ 9 8 6 "$id" "$adapter" "$FM_PROCEVENT_EXTENSION_ID" \ "$FM_PROCEVENT_EXTENSION_VERSION" "$FM_PROCEVENT_EXTENSION_CAPABILITY_VERSION" \ "$FM_PROCEVENT_EXTENSION_PACKAGE_DIGEST" "$FM_PROCEVENT_EXTENSION_BINDING_DIGEST" \ - "$CLAIM_TOKEN" "$runner" "$out" "$$" "$(fm_pid_identity "$$")" "$MAX_OUTPUT_BYTES" -- "${ARGV[@]}") \ - || die "cannot safely stage the extension result" + "$CLAIM_TOKEN" "$runner" "$out" "$$" "$(fm_pid_identity "$$")" "$MAX_OUTPUT_BYTES" \ + "$launch_ready" -- "${ARGV[@]}" > "$launch_reply" & + launch_pid=$! + while [ ! -s "$REG/$launch_ready" ] && kill -0 "$launch_pid" 2>/dev/null; do sleep 0.01; done + fm_procevent_source_lock_release "$id" \ + || die "cannot release the source launch boundary: $id" + wait "$launch_pid" || { + rm -f -- "$REG/$launch_ready" "$launch_reply" + die "cannot safely stage the extension result" + } + [ -s "$REG/$launch_ready" ] || { + rm -f -- "$REG/$launch_ready" "$launch_reply" + die "cannot establish the source launch boundary: $id" + } + IFS= read -r capture_state < "$launch_reply" || capture_state= + rm -f -- "$REG/$launch_ready" "$launch_reply" IFS=$'\t' read -r capture_state durable rc truncated reservation_terminal reservation_silent < "$out") || die "cannot stage output" + [ ! -e "$out" ] && [ ! -L "$out" ] || { + fm_procevent_source_lock_release "$id" + die "cannot safely stage output" + } + (umask 077; : > "$out") || { + fm_procevent_source_lock_release "$id" + die "cannot stage output" + } STAGED_OUTPUT=$out - "${ARGV[@]}" 2>/dev/null | perl -e ' + launch_ready="$REG/.$id.$CLAIM_TOKEN.launch-pipe" + mkfifo -m 600 "$launch_ready" || { + fm_procevent_source_lock_release "$id" + die "cannot prepare the source launch boundary: $id" + } + exec 5<> "$launch_ready" || { + rm -f -- "$launch_ready" + fm_procevent_source_lock_release "$id" + die "cannot retain the source launch boundary: $id" + } + exec 4< "$launch_ready" || { + exec 5>&- + rm -f -- "$launch_ready" + fm_procevent_source_lock_release "$id" + die "cannot retain the source output boundary: $id" + } + "${ARGV[@]}" >&5 5>&- 4<&- 2>/dev/null & + launch_pid=$! + exec 5>&- + rm -f -- "$launch_ready" + fm_procevent_source_lock_release "$id" \ + || die "cannot release the source launch boundary: $id" + perl -e ' use strict; use warnings; my $limit = shift; @@ -795,10 +938,11 @@ EOF $truncated = 1 if $take < $count; } exit($truncated ? 3 : 0); - ' "$MAX_OUTPUT_BYTES" > "$out" - local pipe_status=("${PIPESTATUS[@]}") - rc=${pipe_status[0]} - bound_rc=${pipe_status[1]} + ' "$MAX_OUTPUT_BYTES" <&4 > "$out" + bound_rc=$? + exec 4<&- + wait "$launch_pid" + rc=$? case "$bound_rc" in 0) ;; 3) truncated=1 ;; @@ -923,6 +1067,106 @@ retire_owned_terminal_source() { # return "$status" } +# Bind this runner's lifetime to the home that owns it. Started once the +# claim is held, so the guard names the exact generation it protects, and +# detached into its OWN process group so the group signal it may later send +# reaches the runner and every descendant without killing the guard first. +# If signalling cannot be proved safe or does not finish, the guard remains +# alive and retries on its normal check cadence rather than abandoning cleanup. +start_owner_guard() { # + local identity ready value + identity=$(fm_pid_identity "$$" 2>/dev/null) || return 1 + ready=$(umask 077; mktemp "$REG/.owner-guard-ready.XXXXXX") || return 1 + if ! isolate_process detach "$SCRIPT_DIR/fm-procevent.sh" _owner-watchdog \ + "$1" "$$" "$identity" "$ready" "$CLAIM_STATE_DEVICE" "$CLAIM_STATE_INODE"; then + rm -f -- "$ready" + return 1 + fi + for _ in $(seq 1 50); do + if [ -s "$ready" ]; then + IFS= read -r value < "$ready" || value= + rm -f -- "$ready" + [ "$value" = ready ] + return $? + fi + sleep 0.1 + done + rm -f -- "$ready" + return 1 +} + +# The runner's owner guard, which bounds an accidentally orphaned detached +# runner after its home ends. It revalidates the recorded physical state root +# and its lease on a bounded cadence and, after two consecutive checks cannot prove +# both, invokes the identity-gated stop for the runner's whole process group - +# which is what reaches the blocking child and everything that child spawned, +# exactly as retirement does. A failed verified stop stays on the retry cadence; +# an absent leader ends the guard without signalling an ambiguous group. +# +# Scope is the owning state root and this one runner generation. It never +# matches on a script name, a command line, or a process name: those are shared +# by every home running the same adapter, and a live source in another home +# proves its own owner through that home's own lease. +cmd_owner_watchdog() { # + local id=${1-} pid=${2-} identity=${3-} ready=${4-} state_device=${5-} state_inode=${6-} + local lease tick misses=0 pid_state state_identity current_device current_inode + [ "$#" -eq 6 ] || usage + fm_procevent_source_id_valid "$id" || die "source id must be path-safe: $id" + case "$pid" in ''|*[!0-9]*) die "runner pid must be a positive integer: $pid" ;; esac + [ -n "$identity" ] || die "runner identity is required" + case "$state_device" in ''|*[!0-9]*) die "state device must be an integer" ;; esac + case "$state_inode" in ''|*[!0-9]*) die "state inode must be an integer" ;; esac + [ "${ready%/*}" = "$REG" ] && [ -f "$ready" ] && [ ! -L "$ready" ] \ + || die "owner guard readiness boundary is invalid" + trap 'printf "failed\n" > "$ready" 2>/dev/null || true' EXIT + require_isolated_group guard + lease=$(fm_procevent_owner_lease_seconds) \ + || die "FM_PROCEVENT_OWNER_LEASE_SECONDS must be whole seconds from $FM_PROCEVENT_OWNER_LEASE_MIN_SECONDS to $FM_PROCEVENT_OWNER_LEASE_MAX_SECONDS" + tick=$(fm_procevent_owner_check_seconds) \ + || die "FM_PROCEVENT_OWNER_CHECK_SECONDS must be whole seconds from $FM_PROCEVENT_OWNER_CHECK_MIN_SECONDS to $FM_PROCEVENT_OWNER_CHECK_MAX_SECONDS" + fm_procevent_pid_state "$pid" "$identity" + pid_state=$? + [ "$pid_state" -eq 0 ] || die "runner identity changed before owner guard initialization" + state_identity=$(fm_procevent_claim_state_root_identity "$STATE") \ + || die "owning state root identity is unreadable at owner guard initialization" + IFS=$'\t' read -r _ current_device current_inode _ _ <<< "$state_identity" + [ "$current_device" = "$state_device" ] && [ "$current_inode" = "$state_inode" ] \ + || die "owning state root identity changed before owner guard initialization" + fm_procevent_owner_alive "$STATE" "$lease" \ + || die "owning home lease is not fresh at owner guard initialization" + printf 'ready\n' > "$ready" || die "cannot confirm owner guard initialization" + trap - EXIT + while :; do + sleep "$tick" + fm_procevent_pid_state "$pid" "$identity" + pid_state=$? + case "$pid_state" in + 1|3) exit 0 ;; + 0) ;; + *) continue ;; + esac + state_identity=$(fm_procevent_claim_state_root_identity "$STATE" 2>/dev/null || true) + current_device= + current_inode= + [ -z "$state_identity" ] \ + || IFS=$'\t' read -r _ current_device current_inode _ _ <<< "$state_identity" + if [ "$current_device" = "$state_device" ] \ + && [ "$current_inode" = "$state_inode" ] \ + && fm_procevent_owner_alive "$STATE" "$lease"; then + misses=0 + continue + fi + # Two consecutive misses, so one unreadable read cannot end a live runner. + misses=$((misses + 1)) + [ "$misses" -ge 2 ] || continue + if stop_runner_pid "$pid" "$identity"; then + exit 0 + fi + # Identity/group inspection and signalling can fail transiently. Keep the + # guard alive so the next normal tick retries the same generation cleanup. + done +} + # Start a runner outside the watcher cycle that noticed it was missing. The # public start boundary establishes its own process group before claiming. detach_runner() { # @@ -931,6 +1175,7 @@ detach_runner() { # cmd_reconcile() { local rec id published started=0 stopped=0 uncertain=0 claim owner pid token identity claim_state stop_state + owner_lease_refresh published=$(publish_pending) # Stop a runner this home owns whose source is no longer registered. Without @@ -1008,30 +1253,8 @@ cmd_reconcile() { uncertain=$((uncertain + 1)) fi elif [ "$claim_state" -eq 3 ]; then - # The leader crashed but its owned group is still consuming the - # source. Never start a replacement alongside it: stop that group and - # release its generation first, and if either cannot be proved, keep - # the claim and retry on a later cycle rather than adding a second - # poller. Only the owning home may signal its own group. - owner=$FM_PROCEVENT_CLAIM_HOME - pid=$FM_PROCEVENT_CLAIM_PID - token=$FM_PROCEVENT_CLAIM_TOKEN - identity=$FM_PROCEVENT_CLAIM_IDENTITY - stop_state=2 - if fm_procevent_claim_owned_by_state "$STATE" "$FM_HOME"; then - stop_runner_pid "$pid" "$identity" - stop_state=$? - fi - if [ "$stop_state" -eq 0 ] \ - && cleanup_extension_registration_invocations_locked "$id" \ - && fm_procevent_claim_reclaim_locked "$id" "$owner" "$pid" "$token" 2>/dev/null; then - rm -f -- "$(staging_file "$id" "$token")" - rm -f -- "$(runner_file "$id")" - fm_procevent_source_lock_release "$id" - detach_runner "$id" - started=$((started + 1)) - continue - fi + # A leaderless group's generation is ambiguous under PID/PGID reuse, + # so preserve its claim without signalling or starting a replacement. uncertain=$((uncertain + 1)) elif [ "$claim_state" -eq 2 ]; then uncertain=$((uncertain + 1)) @@ -1047,38 +1270,40 @@ cmd_reconcile() { # its own process group leader, so the group signal is what actually reaches the # blocking child - signalling only the runner would leave that child alive and # reparented, which is exactly how a source that never completes leaks. -stop_runner_pid() { # - local pid=${1-} identity=${2-} state pgid i=0 - case "$pid" in ''|*[!0-9]*) return 2 ;; esac - [ -n "$identity" ] || return 2 +runner_group_signal() { # + local signal=$1 pid=$2 identity=$3 state pgid + # KNOWN LIMIT: only an alive identity-matched leader proves group ownership. + # Detected reused PIDs and absent leaders are refused before signalling; + # launch pacing, leases, and reconcile cleanup are the backstop. fm_procevent_pid_state "$pid" "$identity" state=$? case "$state" in - 0) - # A live identity-matched leader still owns its group, so prove the group - # really is the one this pid leads before signalling it. - pgid=$(ps -o pgid= -p "$pid" 2>/dev/null | tr -d '[:space:]') || return 2 - [ "$pgid" = "$pid" ] || return 2 - ;; - 3) - # The leader crashed but its owned group is still running. Its pgid cannot - # be read from the dead leader, and it does not need to be: only an absent - # leader reaches this state, so the group cannot belong to a reused pid. - ;; - *) return "$state" ;; + 0) ;; + 1) fm_procevent_group_alive "$pid" && return 2; return 1 ;; + *) return 2 ;; esac - kill -TERM -"$pid" 2>/dev/null || return 2 + pgid=$(ps -o pgid= -p "$pid" 2>/dev/null | tr -d '[:space:]') || return 2 + [ "$pgid" = "$pid" ] || return 2 + # KNOWN LIMIT: portable shell cannot make this verification and signal atomic, + # so the PID and group could be reused in the interval between them. + kill -"$signal" -"$pid" 2>/dev/null || return 2 +} + +stop_runner_pid() { # + local pid=${1-} identity=${2-} signal_state i=0 + case "$pid" in ''|*[!0-9]*) return 2 ;; esac + [ -n "$identity" ] || return 2 + runner_group_signal TERM "$pid" "$identity" + signal_state=$? + [ "$signal_state" -eq 0 ] || return "$signal_state" while [ "$i" -lt 20 ]; do kill -0 -"$pid" 2>/dev/null || return 0 - if kill -0 "$pid" 2>/dev/null; then - fm_procevent_pid_state "$pid" "$identity" - state=$? - [ "$state" -eq 2 ] && return 2 - fi sleep 0.1 i=$((i + 1)) done - kill -KILL -"$pid" 2>/dev/null || return 2 + runner_group_signal KILL "$pid" "$identity" + signal_state=$? + [ "$signal_state" -eq 0 ] || return "$signal_state" i=0 while [ "$i" -lt 20 ]; do kill -0 -"$pid" 2>/dev/null || return 0 @@ -1115,6 +1340,7 @@ cmd_handled() { local id=${1-} seq=${2-} status fm_procevent_source_id_valid "$id" || die "source id must be path-safe: $id" case "$seq" in ''|*[!0-9]*) die "sequence must be a nonnegative integer: $seq" ;; esac + owner_lease_refresh fm_procevent_source_lock_acquire "$id" || die "cannot lock source: $id" fm_procevent_mark_handled "$STATE" "$id" "$seq" status=$? @@ -1391,6 +1617,7 @@ cmd_sweep_home() { cmd_list() { local rec id adapter owner pending + owner_lease_refresh if ! fm_procevent_any_registered "$STATE"; then printf 'no sources registered\n' return 0 @@ -1511,6 +1738,7 @@ case "${1-}" in register-extension) shift; cmd_register_extension "$@" ;; start) shift; cmd_start_public "$@" ;; _start) shift; cmd_start "$@" ;; + _owner-watchdog) shift; cmd_owner_watchdog "$@" ;; reconcile) shift; cmd_reconcile "$@" ;; classify) shift; cmd_classify "$@" ;; handled) shift; cmd_handled "$@" ;; diff --git a/bin/fm-remote-file.sh b/bin/fm-remote-file.sh index 34a993db5b4..31887ac27ba 100755 --- a/bin/fm-remote-file.sh +++ b/bin/fm-remote-file.sh @@ -77,7 +77,7 @@ snapshot_bounded_file() { # directory_identity() { if [ "$(uname)" = Darwin ]; then - stat -f '%d:%i' . 2>/dev/null + /usr/bin/stat -f '%d:%i' . 2>/dev/null else stat -c '%d:%i' . 2>/dev/null fi diff --git a/bin/fm-remote-inherit-push.sh b/bin/fm-remote-inherit-push.sh index 189e0629dae..518e849b762 100755 --- a/bin/fm-remote-inherit-push.sh +++ b/bin/fm-remote-inherit-push.sh @@ -27,7 +27,7 @@ sha256_file() { if command -v shasum >/dev/null 2>&1; then shasum -a 256 "$1" | awk '{print $1}'; else sha256sum "$1" | awk '{print $1}'; fi } file_link_count() { - if [ "$(uname)" = Darwin ]; then stat -f %l "$1" 2>/dev/null; else stat -c %h "$1" 2>/dev/null; fi + if [ "$(uname)" = Darwin ]; then /usr/bin/stat -f %l "$1" 2>/dev/null; else stat -c %h "$1" 2>/dev/null; fi } shared_captain_header_valid() { local head diff --git a/bin/fm-remote-inherit.sh b/bin/fm-remote-inherit.sh index be995d75c70..15bb0d4cb1c 100755 --- a/bin/fm-remote-inherit.sh +++ b/bin/fm-remote-inherit.sh @@ -22,7 +22,7 @@ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" die() { printf 'error: %s\n' "$1" >&2; exit 1; } usage() { sed -n '2,10p' "$0" | sed 's/^# \{0,1\}//'; exit 2; } file_link_count() { - if [ "$(uname)" = Darwin ]; then stat -f %l "$1" 2>/dev/null; else stat -c %h "$1" 2>/dev/null; fi + if [ "$(uname)" = Darwin ]; then /usr/bin/stat -f %l "$1" 2>/dev/null; else stat -c %h "$1" 2>/dev/null; fi } sha256_file() { if command -v shasum >/dev/null 2>&1; then shasum -a 256 "$1" | awk '{print $1}'; else sha256sum "$1" | awk '{print $1}'; fi diff --git a/bin/fm-remote-job-lib.sh b/bin/fm-remote-job-lib.sh index f6ac2ad9b99..22e42a4b4ab 100755 --- a/bin/fm-remote-job-lib.sh +++ b/bin/fm-remote-job-lib.sh @@ -758,7 +758,7 @@ fm_remote_job_reap() { # ; only removes an exact completed re fm_remote_job_path_mtime() { # # The platform override controls worker shape in isolated tests, not the host # kernel's stat syntax. - if [ "$(uname -s 2>/dev/null || true)" = Darwin ]; then stat -f %m "$1" 2>/dev/null; else stat -c %Y "$1" 2>/dev/null; fi + if [ "$(uname -s 2>/dev/null || true)" = Darwin ]; then /usr/bin/stat -f %m "$1" 2>/dev/null; else stat -c %Y "$1" 2>/dev/null; fi } fm_remote_job_stage_owner_alive() { # diff --git a/bin/fm-spawn.sh b/bin/fm-spawn.sh index 1c27df82883..03d423028d2 100755 --- a/bin/fm-spawn.sh +++ b/bin/fm-spawn.sh @@ -289,6 +289,11 @@ # and every refusal; a failed registration stops this spawn rather than launching # a worker that would wedge on the dialog. A --secondmate launch never runs it, # so a claude secondmate home keeps its own one-time trust decision. +# Every claude launch also carries the attribution-off policy in its per-launch +# --settings JSON, so a spawned worker never writes a Co-Authored-By trailer, +# Claude-Session link, or generated-with line into a commit or PR body; +# launch_template() below owns the reason it cannot come from the captain's own +# settings. # Publishing the record and moving this home's backlog item to In flight are one # step, not two: bin/fm-backlog-transition-lib.sh owns that invariant, and this # script performs the transition under the task's own meta lock before it reports @@ -1415,7 +1420,15 @@ launch_template() { # alone disables the feature; keep both so a managed override of one still # leaves the other in force. Both are per-launch, scoped to this invocation only, # and never touch the captain's global ~/.claude/settings.json. - claude) printf '%s' 'CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false CLAUDE_CODE_SEND_FEEDBACK=0 claude --dangerously-skip-permissions --settings '\''{"feedbackDrafts":"off"}'\'' __MODELFLAG____EFFORTFLAG__"$(__OPINPUT__ encode launch-brief < __BRIEF__)"' ;; + # The same inline --settings JSON also carries the attribution policy + # ("attribution": {"commit": "", "pr": "", "sessionUrl": false}), which + # suppresses Claude Code's Co-Authored-By trailer, Claude-Session link, and + # generated-with line in commits and PR bodies. The captain sets that + # policy in the `user` settings scope, but a launched worker's settings + # sources are not guaranteed to load that scope, so a worker would + # otherwise run with attribution back on; carrying it per launch keeps the + # policy in force regardless of which settings scopes end up loaded. + claude) printf '%s' 'CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false CLAUDE_CODE_SEND_FEEDBACK=0 claude --dangerously-skip-permissions --settings '\''{"feedbackDrafts":"off","attribution":{"commit":"","pr":"","sessionUrl":false}}'\'' __MODELFLAG____EFFORTFLAG__"$(__OPINPUT__ encode launch-brief < __BRIEF__)"' ;; codex) if [ "$kind" = secondmate ]; then printf '%s' 'codex __MODELFLAG____EFFORTFLAG__--dangerously-bypass-approvals-and-sandbox "$(__OPINPUT__ encode launch-brief < __BRIEF__)"' diff --git a/bin/fm-startup-memory-budget-lib.sh b/bin/fm-startup-memory-budget-lib.sh index f2c06014b8e..033bb69ba42 100644 --- a/bin/fm-startup-memory-budget-lib.sh +++ b/bin/fm-startup-memory-budget-lib.sh @@ -23,7 +23,7 @@ fm_startup_memory_budget_fail() { fm_startup_memory_budget_link_count() { if [ "$(uname)" = Darwin ]; then - stat -f %l "$1" 2>/dev/null + /usr/bin/stat -f %l "$1" 2>/dev/null else stat -c %h "$1" 2>/dev/null fi diff --git a/bin/fm-supervise-daemon.sh b/bin/fm-supervise-daemon.sh index 91174bc5baf..a0145067528 100755 --- a/bin/fm-supervise-daemon.sh +++ b/bin/fm-supervise-daemon.sh @@ -236,7 +236,7 @@ _state_root() { printf '%s' "${FM_STATE_OVERRIDE:-$FM_HOME/state}"; } # --- portable stat (same trap as fm-watch.sh: no `stat -f || stat -c`) ------- if [ "$(uname)" = Darwin ]; then - _stat_file_mtime() { stat -f %m "$1" 2>/dev/null; } + _stat_file_mtime() { /usr/bin/stat -f %m "$1" 2>/dev/null; } else _stat_file_mtime() { stat -c %Y "$1" 2>/dev/null; } fi diff --git a/bin/fm-supervision-lib.sh b/bin/fm-supervision-lib.sh index 1705dcc2457..1bbc5708834 100644 --- a/bin/fm-supervision-lib.sh +++ b/bin/fm-supervision-lib.sh @@ -15,7 +15,7 @@ # Portable mtime; Linux stat lacks -f, macOS stat lacks -c. fm_sup_stat_mtime() { if [ "$(uname)" = Darwin ]; then - stat -f %m "$1" 2>/dev/null + /usr/bin/stat -f %m "$1" 2>/dev/null else stat -c %Y "$1" 2>/dev/null fi diff --git a/bin/fm-teardown.sh b/bin/fm-teardown.sh index 99597cded73..eb4182d7a6c 100755 --- a/bin/fm-teardown.sh +++ b/bin/fm-teardown.sh @@ -22,6 +22,9 @@ # The close - and only the close - is replaced by `tasks-axi reopen` with the # deliverable recorded while the backlog item is still an open captain call # (bin/fm-captain-hold.sh `open` owns that predicate), because the policy holds +# NOTE: this uses `open`'s silent default and depends only on its unchanged +# 0/1/2 exit-code contract. The optional `--identity` output that bin/fm-watch.sh +# asks for prints only on an exit 0 and changes nothing read here. # the very work item a question gates and cleanup must never retire the # captain's own question. The same pending-close record carries that intent as # `mode=retain`, so an interrupted cleanup replays the retention rather than a diff --git a/bin/fm-test-isolation-proof.sh b/bin/fm-test-isolation-proof.sh index e9ecd53d32d..4ed6175ae38 100755 --- a/bin/fm-test-isolation-proof.sh +++ b/bin/fm-test-isolation-proof.sh @@ -216,8 +216,8 @@ EOF dir_mode() { local path=$1 - if stat -f %Lp "$path" >/dev/null 2>&1; then - stat -f %Lp "$path" + if /usr/bin/stat -f %Lp "$path" >/dev/null 2>&1; then + /usr/bin/stat -f %Lp "$path" else stat -c %a "$path" fi diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 7d576ba670e..b5cdd2964ef 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -2273,7 +2273,7 @@ else cat "$out" fi if worker_root_mode_is_enforceable; then - mode=$(stat -c %a "$work" 2>/dev/null || stat -f %Lp "$work" 2>/dev/null || echo unknown) + mode=$(stat -c %a "$work" 2>/dev/null || /usr/bin/stat -f %Lp "$work" 2>/dev/null || echo unknown) case "$mode" in 700|0700) ;; *) diff --git a/bin/fm-turnend-guard.sh b/bin/fm-turnend-guard.sh index d31e2520170..398fa68b7b7 100755 --- a/bin/fm-turnend-guard.sh +++ b/bin/fm-turnend-guard.sh @@ -35,9 +35,17 @@ # Away mode (state/.afk): the away-mode daemon owns supervision and runs the # watcher one-shot, restarting it after every wake, so the watch lock is # regularly unheld at a turn boundary with nothing wrong. A live -# identity-matched daemon holding this home, plus the unchanged fresh-beacon -# test, is what proves supervision there - see fm_afk_daemon_owns_supervision in -# bin/fm-wake-lib.sh. The strict watcher predicate is unchanged everywhere else. +# identity-matched daemon holding this home, plus a fresh beacon, is what +# proves supervision there - see fm_afk_daemon_owns_supervision in +# bin/fm-wake-lib.sh. The beacon freshness test there uses AFK_GRACE +# (fm_poll_derived_grace, docs/turnend-guard.md "Guard grace and the poll +# cadence"), not the flat $GRACE every other check on this page uses: the +# daemon starts a fresh one-shot watcher only after it finishes handling the +# previous wake, and that handling can legitimately run past a flat 300s +# window under load (a slow registered check, a busy supervisor pane) with the +# daemon perfectly healthy throughout. The strict watcher predicate and $GRACE +# are unchanged everywhere else, including for a dead daemon pid or a beacon +# older than AFK_GRACE, which still block. # # Loop-guard, codex/Grok (default) mode: never block twice in the same turn. # Codex uses stop_hook_active and Grok uses stopHookActive; typed camel-case @@ -191,10 +199,15 @@ fi # hand-off, when no watcher process holds the lock and nothing is wrong, so # requiring one here alarmed on healthy away-mode supervision. A live # identity-matched daemon holding this home is the right owner to test for. -# The beacon half of the predicate is deliberately unchanged: a daemon that -# stops restarting its watcher still blocks once the beacon passes grace, and -# a home with no daemon and no watcher blocks exactly as before. -if [ "$FM_SUP_WATCHER_FRESH" = true ] && fm_afk_daemon_owns_supervision "$STATE"; then +# The beacon half of the predicate still applies: a daemon that stops +# restarting its watcher still blocks once the beacon passes grace, and a home +# with no daemon and no watcher blocks exactly as before. It uses AFK_GRACE +# (poll-cadence-derived, see the comment above) instead of the flat $GRACE +# every other check on this page uses, so a daemon that is genuinely still +# cycling - just slower than a fixed 300s window - is not misread as down. +AFK_GRACE=${FM_GUARD_GRACE:-$(fm_poll_derived_grace)} +if [ "$(fm_path_age "$STATE/.last-watcher-beat")" -lt "$AFK_GRACE" ] \ + && fm_afk_daemon_owns_supervision "$STATE"; then allow_supervised_stop fi diff --git a/bin/fm-wake-drain.sh b/bin/fm-wake-drain.sh index 73bbd30d8c6..8268bb917fa 100755 --- a/bin/fm-wake-drain.sh +++ b/bin/fm-wake-drain.sh @@ -1,5 +1,6 @@ #!/usr/bin/env bash -# Present durable watcher wake records, optionally acknowledge handled records, +# Present durable watcher wake records, retire rows no actor could ever consume, +# optionally acknowledge handled records, # annotate every unread line for validated signal status keys, surface unread # informational status lines, latest captain-facing statuses not covered by a # newer branch outcome, OPEN DECISIONS, and captain-call record divergence, @@ -67,38 +68,71 @@ ELIGIBLE_ROWS_FILE="$STATE/.branch-eligible-rows" ELIGIBLE_OWNER_FILE="$STATE/.branch-eligible-owner" MAIN_ROWS_FILE="$STATE/.main-eligible-rows" -rows_file_valid() { - [ -s "$1" ] && awk 'BEGIN { ok=1 } !/^[0-9]+$/ || seen[$0]++ { ok=0 } END { exit !ok }' "$1" -} - -branch_grant_live_locked() { - local version pid identity generation current - [ -f "$ELIGIBLE_OWNER_FILE" ] && [ ! -L "$ELIGIBLE_OWNER_FILE" ] || return 1 - exec 8< "$ELIGIBLE_OWNER_FILE" || return 1 - IFS= read -r version <&8 || { exec 8<&-; return 1; } - IFS= read -r pid <&8 || { exec 8<&-; return 1; } - IFS= read -r identity <&8 || { exec 8<&-; return 1; } - IFS= read -r generation <&8 || { exec 8<&-; return 1; } - if IFS= read -r _extra <&8; then exec 8<&-; return 1; fi - exec 8<&- - [ "$version" = fm-branch-eligible-owner-v1 ] || return 1 - case "$pid" in ''|*[!0-9]*|1) return 1 ;; esac - case "$generation" in ''|*[!A-Za-z0-9._-]*) return 1 ;; esac - current=$(fm_pid_identity "$pid" 2>/dev/null) || return 1 - [ -n "$current" ] && [ "$current" = "$identity" ] -} +rows_file_valid() { fm_wake_grant_rows_valid "$1"; } reclaim_stale_branch_grant_locked() { [ -e "$ELIGIBLE_ROWS_FILE" ] || [ -L "$ELIGIBLE_ROWS_FILE" ] || return 0 - if ! rows_file_valid "$ELIGIBLE_ROWS_FILE" || ! branch_grant_live_locked; then + if ! fm_wake_branch_grant_live "$ELIGIBLE_ROWS_FILE" "$ELIGIBLE_OWNER_FILE"; then rm -f -- "$ELIGIBLE_ROWS_FILE" "$ELIGIBLE_OWNER_FILE" fi } +# Retire rows no actor can ever consume. A claim, a presentation, and an +# acknowledgement all require the five appended fields and a numeric sequence, +# so a truncated or corrupted row is counted as queued while it can never be +# presented and can never be named by an --ack-through cutoff: left alone it +# wedges the queue for good. Main owns that repair - a branch grant can only +# name sequences that were structurally valid when it was published - and it +# runs under the queue lock, so no concurrent append is observed half-written. +# A repair that cannot be written (state/ full, unwritable, unreadable) is +# reported and never fatal: the usable rows are still presentable and +# acknowledgeable, and failing the whole drain would strand them too. +retire_unconsumable_rows_locked() { + local retired unusable queued kept + [ -f "$FM_WAKE_QUEUE" ] || return 0 + if DRAIN_TMP=$(mktemp "$STATE/.wake-queue.retire.XXXXXX") \ + && chmod 0600 "$DRAIN_TMP" \ + && unusable=$(awk -F '\t' -v keep="$DRAIN_TMP" ' + NF >= 5 && $2 ~ /^[0-9]+$/ { print > keep; next } + { shown++; if (shown <= 20) printf "wake drain: %s\n", $0 } + END { if (shown > 20) printf "wake drain: ... %d further unusable row(s) not shown\n", shown - 20 } + ' "$FM_WAKE_QUEUE"); then + queued=$(awk 'END { print NR }' "$FM_WAKE_QUEUE") + kept=$(awk 'END { print NR }' "$DRAIN_TMP") + retired=$(( queued - kept )) + if [ "$retired" -eq 0 ]; then + rm -f -- "$DRAIN_TMP" + DRAIN_TMP= + return 0 + fi + if _fm_atomic_replace "$DRAIN_TMP" "$FM_WAKE_QUEUE"; then + DRAIN_TMP= + printf 'wake drain: retired %s unusable queue row(s) that carried no sequence to present or acknowledge:\n%s\n' \ + "$retired" "$unusable" >&2 + return 0 + fi + fi + printf 'wake drain: unusable queue row(s) could not be retired (check that %s is readable and %s is writable); continuing with the rows that remain usable\n' \ + "$FM_WAKE_QUEUE" "$STATE" >&2 +} + +# One bounded line naming the rows a live branch grant is holding, so a main +# drain with nothing of its own never looks like a silently swallowed wake. +print_branch_held_notice() { + local held seqs + held=$(fm_wake_actor_pending_count branch "$ELIGIBLE_ROWS_FILE" "$ELIGIBLE_OWNER_FILE") || return 0 + [ "$held" -gt 0 ] || return 0 + seqs=$(fm_wake_grant_rows_valid "$ELIGIBLE_ROWS_FILE" \ + && awk 'NR <= 20 { printf "%s%s", (NR > 1 ? "," : ""), $1 } END { if (NR > 20) printf ",..." }' \ + "$ELIGIBLE_ROWS_FILE") + printf 'WAKE ROWS HELD BY SUPERVISION BRANCH: %s queued row(s) (%s) are granted to the live supervision branch, which presents and acknowledges them.\n' \ + "$held" "${seqs:-unknown}" +} + write_rows_file_locked() { # local target=$1 source=$2 if [ ! -s "$source" ]; then - rm -f -- "$target" + rm -f -- "$target" "$source" return fi chmod 0600 "$source" || return 1 @@ -606,6 +640,7 @@ else fi DRAIN_LOCK_HELD=true reclaim_stale_branch_grant_locked || exit 1 +[ "$ACTOR" != main ] || retire_unconsumable_rows_locked [ "$ACTOR" != branch ] || require_branch_eligible_rows || exit 1 if [ -n "$ACK_THROUGH" ]; then @@ -756,6 +791,11 @@ if [ "$ACTOR" = main ]; then fi claim_main_rows_locked || exit 1 if [ ! -s "$MAIN_ROWS_FILE" ]; then + # Every remaining row is reserved by the live branch grant, which presents + # and acknowledges them itself. Say so rather than exiting silently: a + # drain that prints nothing while the queue is visibly non-empty reads as a + # lost wake, and leaves the caller with no idea who owns what is queued. + print_branch_held_notice fm_lock_release "$FM_WAKE_QUEUE_LOCK" DRAIN_LOCK_HELD=false (print_status_presentation) || true diff --git a/bin/fm-wake-grant.sh b/bin/fm-wake-grant.sh index 2cc604f5f2c..bdff2fd4ead 100755 --- a/bin/fm-wake-grant.sh +++ b/bin/fm-wake-grant.sh @@ -22,27 +22,11 @@ trap cleanup EXIT trap 'exit 130' INT trap 'exit 143' TERM -rows_valid() { - [ -s "$1" ] && awk 'BEGIN { ok=1 } !/^[0-9]+$/ || seen[$0]++ { ok=0 } END { exit !ok }' "$1" -} +# fm-wake-lib.sh owns both the grant row-list shape and the owner-record read. +rows_valid() { fm_wake_grant_rows_valid "$1"; } -owner_matches() { - local expected_pid=${1:-} expected_generation=${2:-} version pid identity generation current - [ -f "$BRANCH_OWNER" ] && [ ! -L "$BRANCH_OWNER" ] || return 1 - exec 8< "$BRANCH_OWNER" || return 1 - IFS= read -r version <&8 || { exec 8<&-; return 1; } - IFS= read -r pid <&8 || { exec 8<&-; return 1; } - IFS= read -r identity <&8 || { exec 8<&-; return 1; } - IFS= read -r generation <&8 || { exec 8<&-; return 1; } - if IFS= read -r _extra <&8; then exec 8<&-; return 1; fi - exec 8<&- - [ "$version" = fm-branch-eligible-owner-v1 ] || return 1 - case "$pid" in ''|*[!0-9]*|1) return 1 ;; esac - case "$generation" in ''|*[!A-Za-z0-9._-]*) return 1 ;; esac - [ -z "$expected_pid" ] || [ "$pid" = "$expected_pid" ] || return 1 - [ -z "$expected_generation" ] || [ "$generation" = "$expected_generation" ] || return 1 - current=$(fm_pid_identity "$pid" 2>/dev/null) || return 1 - [ -n "$current" ] && [ "$current" = "$identity" ] +owner_matches() { # [] [] + fm_wake_branch_owner_matches "$BRANCH_OWNER" "${1:-}" "${2:-}" } case "${1:-}" in diff --git a/bin/fm-wake-lib.sh b/bin/fm-wake-lib.sh index 17d55afdcea..8963d464011 100755 --- a/bin/fm-wake-lib.sh +++ b/bin/fm-wake-lib.sh @@ -92,7 +92,7 @@ fm_pid_identity() { fm_path_mtime() { if [ "$_FM_UNAME" = Darwin ]; then - stat -f %m "$1" 2>/dev/null + /usr/bin/stat -f %m "$1" 2>/dev/null else stat -c %Y "$1" 2>/dev/null fi @@ -104,6 +104,25 @@ fm_path_age() { echo $(( $(date +%s) - m )) } +# fm_poll_derived_grace [poll-seconds] +# Default guard-grace derivation: max(300, poll + 60). A watcher touches its +# liveness beacon once per poll cycle, so a fixed 300s grace stops correctly +# bounding staleness once the poll cadence reaches or exceeds it; growing the +# default with the cadence while keeping the historical 300s floor for the +# common short-poll case fixes that without a caller-specific constant. +# Defaults to $FM_POLL (fm-watch.sh's own poll env var) when no argument is +# given, so a caller with no independent notion of the poll cadence still +# derives the same default fm-watch.sh itself would use. +# docs/turnend-guard.md "Guard grace and the poll cadence" is the single owner +# of the rationale; every FM_GUARD_GRACE default should derive from this. +fm_poll_derived_grace() { + local poll=${1:-${FM_POLL:-15}} margin=60 derived + case "$poll" in ''|*[!0-9]*) poll=15 ;; esac + derived=$((poll + margin)) + [ "$derived" -ge 300 ] || derived=300 + printf '%s\n' "$derived" +} + # fm_watcher_lock_unheld # True when the watcher lock or its symlinked owner directory is absent, or when # the existing lock records no pid at all. Any non-empty pid remains held here; @@ -1649,6 +1668,23 @@ fm_wake_queued_keys_locked() { "$FM_WAKE_QUEUE" 2>/dev/null || true } +fm_wake_secondmate_progress_marker_write() { # + local task=$1 observed_at=$2 oldest_row_key=$3 marker tmp + case "$task" in ''|*[!A-Za-z0-9._-]*) return 1 ;; esac + case "$observed_at" in ''|*[!0-9]*) return 1 ;; esac + case "$oldest_row_key" in ''|*[!0-9-]*) return 1 ;; esac + marker="$STATE/.secondmate-wake-progress-$task" + if [ -e "$marker" ] || [ -L "$marker" ]; then + [ -f "$marker" ] && [ ! -L "$marker" ] || return 1 + fi + tmp=$(mktemp "$STATE/.secondmate-wake-progress.XXXXXX") || return 1 + if ! printf '%s\t%s\n' "$observed_at" "$oldest_row_key" > "$tmp" || ! chmod 0600 "$tmp" \ + || ! _fm_atomic_replace "$tmp" "$marker"; then + rm -f -- "$tmp" + return 1 + fi +} + fm_wake_secondmate_stall_marker_write() { # local task=$1 row_key=$2 marker tmp case "$task" in ''|*[!A-Za-z0-9._-]*) return 1 ;; esac @@ -1746,6 +1782,86 @@ fm_wake_print_deduped() { ' "$file" } +# --- branch grant evidence and per-actor pending rows ------------------------ +# +# docs/watcher-continuity.md "Per-actor acknowledgement" owns the contract these +# helpers read; this is its single implementation, shared by the drain (which +# repairs and consumes a grant under the queue lock), the grant publisher, and +# the guard (which only counts, and never takes the lock). + +# 0 when is a non-empty list of distinct sequence numbers. +fm_wake_grant_rows_valid() { # + [ -s "$1" ] && awk 'BEGIN { ok=1 } !/^[0-9]+$/ || seen[$0]++ { ok=0 } END { exit !ok }' "$1" +} + +# 0 when holds the supported record, names a live process whose +# identity still matches what was recorded, and matches any expected pid and +# generation the caller pins. An unreadable, malformed, or superseded record is +# not a match, so uncertainty reads as "no live owner". +fm_wake_branch_owner_matches() { # [] [] + local file=$1 expected_pid=${2:-} expected_generation=${3:-} + local version pid identity generation current extra + [ -f "$file" ] && [ ! -L "$file" ] || return 1 + exec 8< "$file" || return 1 + IFS= read -r version <&8 || { exec 8<&-; return 1; } + IFS= read -r pid <&8 || { exec 8<&-; return 1; } + IFS= read -r identity <&8 || { exec 8<&-; return 1; } + IFS= read -r generation <&8 || { exec 8<&-; return 1; } + if IFS= read -r extra <&8; then exec 8<&-; return 1; fi + exec 8<&- + [ "$version" = fm-branch-eligible-owner-v1 ] || return 1 + case "$pid" in ''|*[!0-9]*|1) return 1 ;; esac + case "$generation" in ''|*[!A-Za-z0-9._-]*) return 1 ;; esac + [ -z "$expected_pid" ] || [ "$pid" = "$expected_pid" ] || return 1 + [ -z "$expected_generation" ] || [ "$generation" = "$expected_generation" ] || return 1 + current=$(fm_pid_identity "$pid" 2>/dev/null) || return 1 + [ -n "$current" ] && [ "$current" = "$identity" ] +} + +# 0 when a branch grant is currently reserving rows: a valid row snapshot whose +# recorded owner is still live. Anything else means no row is reserved. +fm_wake_branch_grant_live() { # + fm_wake_grant_rows_valid "$1" && fm_wake_branch_owner_matches "$2" +} + +# How many queued rows can act on right now - exactly the rows a drain +# by that actor would present or retire, and therefore the only rows worth +# telling that actor to drain. Main owns every structurally valid row a live +# branch grant does not reserve, plus every structurally invalid row. The branch +# owns exactly the rows its live grant names. Read without the queue lock: a +# torn read can only mis-count one poll, and the drain re-derives the set under +# the lock before it presents or mutates anything. +fm_wake_actor_pending_count() { # [ ] + local actor=${1:-main} rows=${2:-$STATE/.branch-eligible-rows} + local owner=${3:-$STATE/.branch-eligible-owner} grant='' count='' + [ -f "$FM_WAKE_QUEUE" ] || { printf '0\n'; return 0; } + if fm_wake_branch_grant_live "$rows" "$owner"; then + grant=$rows + fi + if [ "$actor" = branch ]; then + [ -n "$grant" ] || { printf '0\n'; return 0; } + count=$(awk -F '\t' -v seqs="$grant" ' + BEGIN { while ((getline line < seqs) > 0) keep[line] = 1 } + NF >= 5 && $2 ~ /^[0-9]+$/ && ($2 in keep) { n++ } + END { print n + 0 } + ' "$FM_WAKE_QUEUE") || count='' + else + count=$(awk -F '\t' -v seqs="$grant" ' + BEGIN { if (seqs != "") while ((getline line < seqs) > 0) reserved[line] = 1 } + NF < 5 || $2 !~ /^[0-9]+$/ { n++; next } + !($2 in reserved) { n++ } + END { print n + 0 } + ' "$FM_WAKE_QUEUE") || count='' + fi + # A queue that exists but cannot be counted (unreadable file, unreadable + # state/) is not evidence of an empty queue: report a pending row so callers + # still raise the alarm on a queue nobody can prove is drained. A failed count + # is decided by awk's exit status, not by what it printed, because an awk that + # reaches END after failing to open the queue would otherwise report 0 rows. + case "$count" in ''|*[!0-9]*) count=1 ;; esac + printf '%s\n' "$count" +} + # --- signal announcement signatures ----------------------------------------- # # The watcher's per-file signal scan (bin/fm-watch.sh scan_signals) detects a @@ -1763,7 +1879,7 @@ fm_wake_signal_sig() { # -> reported-state signature status_observed_signature "$1" ;; *) - if [ "$_FM_UNAME" = Darwin ]; then stat -f '%z:%Fm' "$1" 2>/dev/null; else stat -c '%s:%Y' "$1" 2>/dev/null; fi + if [ "$_FM_UNAME" = Darwin ]; then /usr/bin/stat -f '%z:%Fm' "$1" 2>/dev/null; else stat -c '%s:%Y' "$1" 2>/dev/null; fi ;; esac } diff --git a/bin/fm-watch.sh b/bin/fm-watch.sh index 435383c8bdb..2a802d10d27 100755 --- a/bin/fm-watch.sh +++ b/bin/fm-watch.sh @@ -79,11 +79,14 @@ # check: inactive-outcome bounded poll-loop reconciliation found a suspicious # inactive terminal outcome that still lacks its durable # upstream receipt -# check: secondmate wake-loop stalled: mate= row= age=s -# the oldest valid row in an endpoint-recorded local -# secondmate home's durable wake queue exceeded -# FM_SECONDMATE_WAKE_STALL_SECS; observation is read-only -# and one parent receipt suppresses repeats for that row +# check: secondmate wake-loop stalled: mate= row= idle=s +# an actionable row in an endpoint-recorded local +# secondmate home's durable wake queue did not advance +# between observations for FM_SECONDMATE_WAKE_STALL_SECS +# while the mate was not in an active turn; declared +# external-wait pause rows do not feed this escalation, +# observation is read-only, and one parent notification +# covers each no-progress episode # For normal supervision, resume the session-start primary-harness protocol # after each printed reason. Direct duplicate invocations of this script still # no-op through the watcher singleton lock. @@ -138,7 +141,6 @@ mkdir -p "$STATE" WATCH_LOCK="$STATE/.watch.lock" WATCH_PATH="$SCRIPT_DIR/fm-watch.sh" WATCHER_DOWNTIME_MARKER="$STATE/.watcher-down" -WATCHER_STALE_GRACE=${FM_WATCHER_STALE_GRACE:-${FM_GUARD_GRACE:-300}} # The singleton-lock acquisition, EXIT trap, and the blocking supervision loop # all live below the source guard at the very bottom of this file (see "Main # entry"). Sourcing this file for unit tests therefore loads the functions - @@ -153,8 +155,10 @@ WATCHER_STALE_GRACE=${FM_WATCHER_STALE_GRACE:-${FM_GUARD_GRACE:-300}} # appended to that garbage. Arithmetic under `set -u` then aborts on the stray # token (e.g. the word "File" read as an unset variable), which silently kills the # watcher mid-cycle. Detect the platform once and pick the right form. +# On Darwin, call /usr/bin/stat rather than PATH-resolved stat so GNU coreutils +# cannot shadow the BSD `-f` syntax. if [ "$(uname)" = Darwin ]; then - stat_mtime() { stat -f %m "$1" 2>/dev/null; } # epoch seconds of mtime + stat_mtime() { /usr/bin/stat -f %m "$1" 2>/dev/null; } # epoch seconds of mtime else stat_mtime() { stat -c %Y "$1" 2>/dev/null; } fi @@ -163,6 +167,15 @@ fi # turn-ended signature, annotation staleness checks, and guarded bookkeeping writes. POLL=${FM_POLL:-15} # seconds between cycles +# The liveness beacon is touched once per cycle, immediately before the +# terminal wait below (event_wait_or_sleep) as well as at the top of the next +# one, so a healthy cycle's beacon can legitimately age up to POLL seconds +# between touches. fm_poll_derived_grace (bin/fm-wake-lib.sh, already sourced +# transitively above) is the single owner of the max(300, poll+60) +# derivation - see docs/turnend-guard.md "Guard grace and the poll cadence". +# This recomputes the library default above now that the real configured +# POLL is known. +WATCHER_STALE_GRACE=${FM_WATCHER_STALE_GRACE:-${FM_GUARD_GRACE:-$(fm_poll_derived_grace "$POLL")}} HEARTBEAT=${FM_HEARTBEAT:-600} # base seconds between heartbeat scans HEARTBEAT_MAX=${FM_HEARTBEAT_MAX:-7200} # heartbeat backoff cap CHECK_INTERVAL=${FM_CHECK_INTERVAL:-300} # seconds between *.check.sh sweeps @@ -216,8 +229,12 @@ STALE_ESCALATE_SECS=${FM_STALE_ESCALATE_SECS:-240} # idle secs before a provabl # between completed turns, including long tool calls, builds, or test runs. BUSY_TURN_MAX_SECS=${FM_BUSY_TURN_MAX_SECS:-3600} # A local secondmate's foreign queue is checked on every poll, but only after this -# bounded age can it produce a parent notification. -SECONDMATE_WAKE_STALL_SECS=${FM_SECONDMATE_WAKE_STALL_SECS:-60} +# bounded interval with no drain progress can it produce a parent notification. +# A healthy mate drains its queue between turns, not inside one, so this default +# sits above a real turn; it is only the backstop behind the active-turn gate in +# secondmate_wake_stall_tick, never a substitute for it. +SECONDMATE_WAKE_STALL_SECS=${FM_SECONDMATE_WAKE_STALL_SECS:-} +case "$SECONDMATE_WAKE_STALL_SECS" in ''|*[!0-9]*|0) SECONDMATE_WAKE_STALL_SECS=180 ;; esac # A crew that declared a pause is idling on a known external wait, so its stale # pane is absorbed rather than wedge-escalated. # A captain-held or paused crew whose agent has confidently exited uses the same @@ -632,14 +649,23 @@ recorded_windows() { done } -# Print the oldest structurally valid row in a local secondmate's foreign queue. -# This is a read-only observation: the receiving home owns acknowledgement and -# this parent never changes the row or the foreign queue. +# Print the oldest structurally valid ACTIONABLE row in a local secondmate's +# foreign queue. A stale recheck that explicitly identifies itself as a declared +# external-wait pause is not evidence that the mate's wake loop is stuck: the +# pause cadence already owns that bounded visibility, and blocked waits remain +# actionable because they do not carry this declaration. This is a read-only +# observation: the receiving home owns acknowledgement and this parent never +# changes the row or the foreign queue. secondmate_oldest_queue_row() { # local queue=$1 [ -f "$queue" ] && [ ! -L "$queue" ] || return 0 awk -F '\t' ' - NF >= 5 && $1 ~ /^[0-9]+$/ && $2 ~ /^[0-9]+$/ { + function declared_external_pause(kind, payload) { + return kind == "stale" \ + && payload ~ /^stale: .*\(paused [0-9]+s, awaiting external - declared (pause,|paused\))/ + } + NF >= 5 && $1 ~ /^[0-9]+$/ && $2 ~ /^[0-9]+$/ \ + && !declared_external_pause($3, $5) { if (!found || $2 < seq) { found = 1 seq = $2 @@ -650,14 +676,39 @@ secondmate_oldest_queue_row() { # ' "$queue" 2>/dev/null || true } -# Surface one durable parent check for one unchanged foreign row after its -# bounded age. The primary marker and queued-key check make repeated watcher -# cycles converge without a notification storm, while an empty queue removes -# only this home's marker so a later row can be observed. +# 0 iff is demonstrably inside an active turn, through the watcher's own +# busy-state knowledge: an exact busy verdict from the semantic contract, bounded +# by the same BUSY_TURN_MAX_SECS that stops a busy pane from proving liveness +# forever. A mate mid-turn has not stopped draining its queue - it simply drains +# between turns - so this gate, not the elapsed interval, is what separates a +# healthy mate from a frozen wake loop. Any absence of proof (no window, a failed +# capture, an idle or unknown verdict, a busy pane past the bound) is NOT an +# active turn, so a frozen queue still escalates. +secondmate_in_active_turn() { # + local task=$1 w=$2 tail40 + [ -n "$w" ] || return 1 + ! busy_turn_over_age "$task" || return 1 + tail40=$(fm_backend_capture "$(window_backend "$w")" "$w" 40 "$(window_label "$w")" 2>/dev/null) || return 1 + window_is_busy "$w" "$tail40" +} + +# Surface one durable parent check when the foreign queue's drain position has +# not moved for the bounded interval. The progress marker records that position +# as the same epoch-sequence row identity the stall receipts use, so the timer +# restarts whenever a different row becomes the oldest actionable one - as the +# mate drains, and as a queue reprovisioned under the same task id starts its +# own generation of rows at whatever sequence it restarts, and neither is a +# continued no-progress episode; row creation time belongs to that identity but +# never to the interval. A moved position ends an alerted episode and starts a +# new observation interval, so a newly-oldest row cannot alert immediately while +# a later genuine freeze remains visible. A mate demonstrably inside an active +# turn never escalates, so the interval is only the backstop behind that gate. +# Receipts close the append-before-marker crash window without changing the +# foreign queue. secondmate_wake_stall_tick() { local now=$(( $(date +%s) )) threshold=$SECONDMATE_WAKE_STALL_SECS - local meta task kind remote_host home queue row epoch seq row_key marker receipt receipt_dir notify_key queued age reason - case "$threshold" in ''|*[!0-9]*|0) threshold=60 ;; esac + local meta task kind remote_host home queue row epoch seq row_key marker progress_marker progress observed_at observed_key + local receipt receipt_dir notify_key queued idle reason episode_alerted # Endpoint metadata admits this queue-loop check; secondmate-liveness owns registered mates whose endpoint is missing or dead. for meta in "$STATE"/*.meta; do [ -e "$meta" ] || continue @@ -675,9 +726,10 @@ secondmate_wake_stall_tick() { queue="$home/state/.wake-queue" row=$(secondmate_oldest_queue_row "$queue") marker="$STATE/.secondmate-wake-stall-$task" + progress_marker="$STATE/.secondmate-wake-progress-$task" receipt_dir="$STATE/.secondmate-wake-stall-receipts/$task" if [ -z "$row" ]; then - rm -f "$marker" + rm -f "$marker" "$progress_marker" if [ -e "$receipt_dir" ] || [ -L "$receipt_dir" ]; then [ -d "$receipt_dir" ] && [ ! -L "$receipt_dir" ] || return 1 rm -rf -- "$receipt_dir" || return 1 @@ -689,17 +741,39 @@ $row EOF case "$epoch" in ''|*[!0-9]*) continue ;; esac case "$seq" in ''|*[!0-9]*) continue ;; esac - age=$((now - epoch)) - [ "$age" -ge "$threshold" ] || continue row_key="$epoch-$seq" - receipt="$receipt_dir/$row_key" + episode_alerted=0 if [ -e "$marker" ] || [ -L "$marker" ]; then [ -f "$marker" ] && [ ! -L "$marker" ] || return 1 + episode_alerted=1 + fi + progress=$(cat "$progress_marker" 2>/dev/null || true) + observed_at=${progress%%[[:space:]]*} + observed_key=${progress#*[[:space:]]} + if [ "$observed_at" = "$progress" ]; then + observed_key= + else + observed_key=${observed_key%%[[:space:]]*} + fi + case "$observed_at" in ''|*[!0-9]*) observed_at= ;; esac + case "$observed_key" in ''|*[!0-9-]*) observed_key= ;; esac + if [ -z "$observed_at" ] || [ -z "$observed_key" ] \ + || [ "$now" -lt "$observed_at" ] || [ "$row_key" != "$observed_key" ]; then + fm_wake_secondmate_progress_marker_write "$task" "$now" "$row_key" || return 1 + [ "$episode_alerted" -eq 0 ] || rm -f "$marker" || return 1 + continue + fi + [ "$episode_alerted" -eq 0 ] || continue + idle=$((now - observed_at)) + [ "$idle" -ge "$threshold" ] || continue + ! secondmate_in_active_turn "$task" "$(fm_backend_target_of_meta "$meta")" || continue + receipt="$receipt_dir/$row_key" + if [ "$(cat "$receipt" 2>/dev/null || true)" = "$row_key" ]; then + fm_wake_secondmate_stall_marker_write "$task" "$row_key" || return 1 + continue fi - [ "$(cat "$marker" 2>/dev/null || true)" = "$row_key" ] && continue - [ "$(cat "$receipt" 2>/dev/null || true)" = "$row_key" ] && continue notify_key="secondmate-wake-loop-$task-$row_key" - reason="check: secondmate wake-loop stalled: mate=$task row=$seq age=${age}s" + reason="check: secondmate wake-loop stalled: mate=$task row=$seq idle=${idle}s" queued=$(fm_wake_queued_keys check) if ! printf '%s\n' "$queued" | grep -Fx "$notify_key" >/dev/null 2>&1; then fm_wake_append check "$notify_key" "$reason" || return 1 @@ -1002,9 +1076,9 @@ pause_state_class() { # # ordinary crew whose agent the gate above confirmed dead, so no live decision gate # is being silenced, or a secondmate, whose endpoint liveness is deliberately never # read and so cannot supply that confirmation. Without the mate case a mate's - # captain hold - which has no current-state mapping and so arrives as `none` - - # would be silenced by every caller rather than taking the bounded re-surface - # cadence, and a forgotten hold would rot invisibly. + # status-declared `captain-held` transfer - which has no current-state mapping + # and so arrives as `none` - would be silenced by every caller rather than taking + # the bounded re-surface cadence, and a forgotten declaration would rot invisibly. [ "$class" = none ] && class=paused case "$class" in paused) date +%s > "$recheck_file" ;; @@ -1013,6 +1087,99 @@ pause_state_class() { # printf '%s' "$class" } +# The two records of one ordinary crew wait, and why its stale alarm reads both. +# +# status_is_paused_or_captain_held reads the status LINE a worker wrote, which is +# the only record when the worker itself is waiting. It is not the only record +# there is: once firstmate hands work to the captain, the wait is written into the +# BACKLOG by bin/fm-captain-hold.sh, and the worker's last line stays whatever it +# was - routinely `done: PR ...` after a delivery, which no line predicate can +# read as a wait. An alarm bounded only by the line therefore re-fires for the +# captain's whole thinking time, on exactly the work they already have in hand. +# +# `open` is that record's own read-only predicate and owns its semantics: exit 0 +# still an open captain call, 1 not, 2 could not be established. Only a 0 bounds +# an alarm here, so an unreadable backlog, an incompatible or absent tasks-axi, +# and a row this home does not carry all keep alarming exactly as they do today - +# a wait this watcher cannot prove is not a wait. +# +# The read costs one subprocess and runs only where the watcher is about to +# alarm, so at most once per distinct stale hash per window, beside the crew-state +# read the same paths already pay. The secondmate stale gate deliberately runs +# before this bound and admits only status-declared waits: a backlog-only hold +# whose mate still says `working:` or `done:` does not reach this read. Reaching +# it would put backlog reads into windows deliberately skipped on ordinary polls. +STALE_WAIT_DECLARATION= + +CAPTAIN_CALL_IDENTITY= + +task_captain_call_open() { # + local task=$1 + CAPTAIN_CALL_IDENTITY= + [ -n "$task" ] || return 1 + CAPTAIN_CALL_IDENTITY=$(FM_HOME="$FM_HOME" "$SCRIPT_DIR/fm-captain-hold.sh" \ + open "$task" --identity 2>/dev/null) || return 1 + return 0 +} + +# The identity a re-surface throttle is bound to: the task's whole status-log +# signature. Any new status event - a replacement wait, a fresh delivery, a +# blocker - changes it and so starts its own window instead of inheriting the +# silence of the one before it. +stale_wait_declaration() { # + printf 'declared:%s' "$(fm_wake_signal_sig "$STATE/$1.status" || true)" +} + +# The same scope for a captain call, carrying the CALL's own lifecycle identity +# beside the status signature. The status log is not enough on its own: a task +# can be answered with `--release` and held again as a genuinely different call +# without any status append, and binding the throttle to the signature alone let +# the second call inherit the first one's silence and absorbed its first sight. +# That first sight is the one alarm this bound must never swallow - a decision +# waiting on the captain that is never surfaced is invisible, where a delivery +# announced twice is merely noise. +captain_call_declaration() { # + printf 'captain-hold:%s:%s' "$2" "$(fm_wake_signal_sig "$STATE/$1.status" || true)" +} + +# 0 when has already been alarmed for this window inside the +# current PAUSE_RESURFACE_SECS. A pure read: recording an alarm is the caller's, +# so the throttle is never advanced by a sighting it just absorbed. +stale_wait_throttled() { # + local throttle="$STATE/.paused-resurfaced-$1" + [ "$(cat "$throttle" 2>/dev/null || true)" = "$2" ] \ + && [ "$(age_of "$throttle")" -lt "$PAUSE_RESURFACE_SECS" ] +} + +# The same bound, for a stale window whose last line IS captain-relevant. That +# line is real and its first sight must still reach the captain, but a delivery +# they are already holding has nothing new to say on the next pane tick. +# Sets STALE_WAIT_DECLARATION to the scope this sighting is bound to, and leaves +# it EMPTY when no open captain call bounds it, so an unheld delivery, a blocker, +# and a failure alarm exactly as they do today. +# Returns 0 to absorb this sighting; 1 to alarm, after which the caller records +# the throttle through stale_wait_record once its own wake append has succeeded. +# Record a fired wake against the bounded cadence, and ONLY after that wake was +# durably appended. A marker written ahead of the append outlives a failed one: +# the watcher exits with no wake queued, and the next sighting reads the fresh +# marker and absorbs the retry, which is the single way this bound could swallow +# an alarm outright rather than delay it. +stale_wait_record() { # + [ -n "$STALE_WAIT_DECLARATION" ] || return 0 + printf '%s' "$STALE_WAIT_DECLARATION" > "$STATE/.paused-resurfaced-$1" +} + +# Bound a due stale alarm for an ordinary crew task held for the captain. +# Backlog-only secondmate holds are outside this guard because the earlier gate +# preserves their no-backlog-read hot path. +captain_call_stale_bound() { # + local key=$1 task=$2 + STALE_WAIT_DECLARATION= + task_captain_call_open "$task" || return 1 + STALE_WAIT_DECLARATION=$(captain_call_declaration "$task" "$CAPTAIN_CALL_IDENTITY") + stale_wait_throttled "$key" "$STALE_WAIT_DECLARATION" +} + # Surface a stale pane no classifier could resolve, so firstmate inspects it: it # may have finished through an interactive menu that wrote no status, be waiting on # a decision, or be wedged. pause_state_class deliberately answers `none` for a @@ -1020,30 +1187,38 @@ pause_state_class() { # # decision is never silenced - which routes every parked-but-live worker here, on # first sight of each distinct stale hash. # -# So a declared wait bounds this path to the same once-per-PAUSE_RESURFACE_SECS +# So a legitimate wait bounds this path to the same once-per-PAUSE_RESURFACE_SECS # cadence resurface_absorbed owns for the absorbed paths, throttled by this # window's own .paused-resurfaced- marker: an idle parked pane still churns # its hash (a clock, a token counter), and each new hash re-enters this path, so -# without that bound one declared wait re-alarms firstmate for its whole duration. +# without that bound one wait re-alarms firstmate for its whole duration. # The FIRST sight still wakes, keeping the inspect-an-inconclusive-state intent, # and the throttle is read BEFORE anything is queued and advanced only by a wake # that really fires - a throttle written by the wake it should have prevented, or # read after that wake was already appended, bounds nothing. +# Both records of an ordinary crew wait bound it (see task_captain_call_open +# above): the status line the worker declared, and the backlog hold firstmate +# recorded once the captain took the work in hand. surface_nonterminal_stale() { # - local win=$1 h=$2 key task last declaration='' declared=1 throttled=1 + local win=$1 h=$2 key task last declared=1 bounded=1 throttled=1 key=$(window_key "$win") task=$(window_to_task "$win" "$STATE") last=$(last_status_line "$STATE/$task.status") + STALE_WAIT_DECLARATION= if status_is_paused_or_captain_held "$last"; then declared=0 - declaration="declared:$(fm_wake_signal_sig "$STATE/$task.status" || true)" - if [ "$(cat "$STATE/.paused-resurfaced-$key" 2>/dev/null || true)" = "$declaration" ] \ - && [ "$(age_of "$STATE/.paused-resurfaced-$key")" -lt "$PAUSE_RESURFACE_SECS" ]; then - throttled=0 - fi + bounded=0 + STALE_WAIT_DECLARATION=$(stale_wait_declaration "$task") + stale_wait_throttled "$key" "$STALE_WAIT_DECLARATION" && throttled=0 + elif captain_call_stale_bound "$key" "$task"; then + bounded=0 + throttled=0 + elif [ -n "$STALE_WAIT_DECLARATION" ]; then + bounded=0 fi if [ "$throttled" -ne 0 ]; then fm_wake_append stale "$win" "stale: $win" || exit 1 + stale_wait_record "$key" fi printf '%s' "$h" > "$STATE/.stale-$key" rm -f "$STATE/.stale-since-$key" @@ -1051,12 +1226,19 @@ surface_nonterminal_stale() { # if [ "$declared" -eq 0 ]; then : > "$STATE/.paused-$key" date +%s > "$STATE/.paused-rechecked-$key" - [ "$throttled" -eq 0 ] || printf '%s' "$declaration" > "$STATE/.paused-resurfaced-$key" + elif [ "$bounded" -eq 0 ]; then + # A backlog hold is NOT a declared pause, and must not be dressed up as one: + # the loop-top reconciliation and pause_state_class both read the status LINE, + # so a .paused-* flag this line does not support would be cleared on the next + # poll - taking the throttle with it - and would hand the mate and dead-agent + # cadences a declaration they were never given. Only the shared re-surface + # marker is kept, which is the whole of what this bound needs. + rm -f "$STATE/.paused-$key" "$STATE/.paused-rechecked-$key" else clear_pause_state "$key" fi if [ "$throttled" -eq 0 ]; then - triage_log "absorbed non-terminal stale (declared wait already re-surfaced this window): $win" + triage_log "absorbed non-terminal stale (declared wait or open captain call already re-surfaced this window): $win" return 0 fi wake "stale: $win" @@ -1874,12 +2056,12 @@ EOF clear_pause_tracking "$key" fi # An idle secondmate endpoint is healthy by design, so a mate is admitted to - # the pane-stale path ONLY to serve a declared wait's bounded re-surface - - # the same declarations pause_state_class reconciles below, which is why this - # gate reads the shared predicate rather than the pause verb alone. Narrowing - # it to `paused` would leave a mate's captain hold rotting invisibly: the - # clear above already spares its pause tracking, but nothing would ever - # re-surface it. + # the pane-stale path ONLY to serve a status-declared wait's bounded + # re-surface. This gate reads the shared predicate rather than the pause verb + # alone so it includes a declared `captain-held` status. A hold recorded only + # in the backlog while the mate still says `working:` or `done:` is outside + # this guard: reaching it would require backlog reads for windows this gate + # deliberately skips, putting that read on the ordinary poll hot path. if [ "$kind" = secondmate ] && ! status_is_paused_or_captain_held "$last"; then continue fi @@ -1937,8 +2119,21 @@ EOF date +%s > "$ssf" clear_write_tracking "$key" triage_log "absorbed stale (provably working, overriding a stale captain-relevant status): $w" + elif captain_call_stale_bound "$key" "$task"; then + # The line is captain-relevant and stays so, but the backlog says + # the captain already holds this work: further NEW pane hashes with + # the same status-log state have nothing to add while they are + # deciding. Only that new-hash repetition is bounded - the first + # sight already alarmed, a new hash inside the window is absorbed, + # and a new hash after it alarms again. A stable hash stays as inert + # here as it already was after a first terminal alarm. + printf '%s' "$h" > "$sf" + rm -f "$ssf" + clear_write_tracking "$key" + triage_log "absorbed stale (open captain call already surfaced for this status): $w" else fm_wake_append stale "$w" "stale: $w" || exit 1 + stale_wait_record "$key" printf '%s' "$h" > "$sf" rm -f "$ssf" clear_write_tracking "$key" diff --git a/bin/fm-x-lib.sh b/bin/fm-x-lib.sh index f50ddce781d..aae910db8cb 100644 --- a/bin/fm-x-lib.sh +++ b/bin/fm-x-lib.sh @@ -89,8 +89,8 @@ fmx_single_link_file_valid() { local file=$1 expected_device=${2-} links device [ -f "$file" ] && [ ! -L "$file" ] || return 1 if [ "$(uname)" = Darwin ]; then - links=$(stat -f %l "$file" 2>/dev/null) || return 1 - device=$(stat -f %d "$file" 2>/dev/null) || return 1 + links=$(/usr/bin/stat -f %l "$file" 2>/dev/null) || return 1 + device=$(/usr/bin/stat -f %d "$file" 2>/dev/null) || return 1 else links=$(stat -c %h "$file" 2>/dev/null) || return 1 device=$(stat -c %d "$file" 2>/dev/null) || return 1 @@ -103,7 +103,7 @@ fmx_single_link_file_mode_valid() { local file=$1 expected_mode=$2 expected_device=${3-} mode fmx_single_link_file_valid "$file" "$expected_device" || return 1 if [ "$(uname)" = Darwin ]; then - mode=$(stat -f %Lp "$file" 2>/dev/null) || return 1 + mode=$(/usr/bin/stat -f %Lp "$file" 2>/dev/null) || return 1 else mode=$(stat -c %a "$file" 2>/dev/null) || return 1 fi @@ -114,8 +114,8 @@ fmx_private_artifact_dir_device() { local dir=$1 mode device [ -d "$dir" ] && [ ! -L "$dir" ] || return 1 if [ "$(uname)" = Darwin ]; then - mode=$(stat -f %Lp "$dir" 2>/dev/null) || return 1 - device=$(stat -f %d "$dir" 2>/dev/null) || return 1 + mode=$(/usr/bin/stat -f %Lp "$dir" 2>/dev/null) || return 1 + device=$(/usr/bin/stat -f %d "$dir" 2>/dev/null) || return 1 else mode=$(stat -c %a "$dir" 2>/dev/null) || return 1 device=$(stat -c %d "$dir" 2>/dev/null) || return 1 @@ -410,7 +410,7 @@ fmx_request_relay_context() { fmx_context_registry_mtime() { local file=$1 mtime - mtime=$(stat -f '%m' "$file" 2>/dev/null) || mtime=$(stat -c '%Y' "$file" 2>/dev/null) || return 1 + mtime=$(/usr/bin/stat -f '%m' "$file" 2>/dev/null) || mtime=$(stat -c '%Y' "$file" 2>/dev/null) || return 1 case "$mtime" in ''|*[!0-9]*) return 1 ;; esac diff --git a/docs/architecture.md b/docs/architecture.md index cff000d5c8b..58b900786d4 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -10,6 +10,12 @@ firstmate's supervisor contract and routing index for conditional procedures is A zero-token bash watcher (`bin/fm-watch.sh`) sleeps on the fleet, classifies detected wakes in bash, and wakes the first mate only when something is actionable. Actionable wakes include captain-relevant status signals, no-verb signals without positive evidence that their crew is still executing, authenticated check output such as PR merge polling or a Relay mention, stale panes whose crew is not provably working whether their status log looks terminal or non-terminal, provably-working stale panes that persist past `FM_STALE_ESCALATE_SECS` without their own task worktree being written, declared external waits and verified captain-held transfers that remain declared past `FM_PAUSE_RESURFACE_SECS`, and heartbeat backstop hits. +For an ordinary crew task, a wait is read from both of its records: the status line a worker declared, and the backlog hold `bin/fm-captain-hold.sh` recorded once firstmate handed the work to the captain. +So a delivered ordinary crew task whose last line stays `done: PR ...` bounds repeated alarms from new pane hashes to the `FM_PAUSE_RESURFACE_SECS` cadence for the length of the captain's decision. +The first hash still alarms, each new hash inside that window is absorbed, and a new hash after the window re-surfaces the hold; a terminal pane hash that never changes stays inert after its first alarm exactly as it did before this bound. +The throttle is scoped to both the current captain-call lifecycle and the status-log state, so releasing and re-holding the same task without a status append starts a fresh window whose first new hash alarms. +A secondmate reaches the stale path only for a wait declared in its status line, so a hold recorded only in the backlog while its last line is `working:` or `done:` is outside this guard. +Reaching that case would require consulting the backlog for windows the secondmate gate deliberately skips, putting backlog reads on the ordinary poll hot path this design preserves. Repeated provably-working stale escalations on the same unchanged pane add an escalation count to the wake reason and, at `FM_WEDGE_DEMAND_INSPECT_COUNT`, a `demand-deep-inspection` marker. A pane holding a file newer than the start of its own quiet window, anywhere in the worktree recorded for that task, is deferred instead of escalated, because a crew writing source, then tests, then documentation behind a static pane is liveness that neither pane quietness nor the run step can show. That deferral re-surfaces on the same `FM_PAUSE_RESURFACE_SECS` cadence as a declared wait, with a reason naming the write evidence rather than a wedge, and it is bounded to one pruned, depth-bounded, wall-clock-bounded walk (`FM_WORKTREE_WRITE_PRUNE`, `FM_WORKTREE_WRITE_MAXDEPTH`, `FM_WORKTREE_WRITE_TIMEOUT`) taken only in the branch that was about to escalate, never on every poll. @@ -21,10 +27,11 @@ Lifting the declaration restores the unchanged busy-pane wedge path, while a pan While away mode is active, a busy pane that crosses the bound under a declared wait is handed to the daemon as the plain wake identity instead of taking that recheck in the watcher, because the daemon owns triage there and a wake already decorated as a possible wedge would override the daemon's own declared-wait verdict; an undeclared busy pane past the bound still takes the wedge escalation in away mode. That handoff is keyed on the declaration itself (the status log's signature) rather than on the pane capture, so a harness footer that ticks on every poll wakes the daemon once per declaration instead of once per poll, and it clears the wedge timer, escalation count, and worktree-write deferral exactly as the normal-mode absorber does, so an undeclared busy phase's timer does not resume when the declaration lifts. Those actionable wakes are written to a durable local queue (`state/.wake-queue`) only after generation-bound recovery evidence is published, so an interrupted watcher or handling turn can be recovered without losing the queue record. -Agent endpoint liveness and queue-consumption liveness are separate: on each poll, the primary watcher reads the oldest valid row from every endpoint-recorded local secondmate home's durable wake queue without locking, consuming, or rewriting that foreign queue. -Once that row reaches `FM_SECONDMATE_WAKE_STALL_SECS`, the primary appends one keyed `check` wake naming the mate, row sequence, and observed age; parent receipts and queued-key deduplication suppress repeats for the same row across watcher and handling crashes, while empty and younger queues remain silent. +Agent endpoint liveness and queue-consumption liveness are separate: on each poll, the primary watcher reads the oldest valid actionable row from every endpoint-recorded local secondmate home's durable wake queue without locking, consuming, or rewriting that foreign queue. +A queue that is draining is not stalled, so the primary times the interval since that oldest actionable row last changed rather than the age of the row itself, and rows that declare themselves a bounded external wait (`awaiting external - declared pause`) are not actionable evidence at all. +Once that no-progress interval reaches `FM_SECONDMATE_WAKE_STALL_SECS` and the mate is not provably inside an active turn (an exact busy verdict, bounded by `FM_BUSY_TURN_MAX_SECS`), the primary appends one keyed `check` wake naming the mate, row sequence, and observed idle interval; parent receipts and queued-key deduplication suppress repeats across watcher and handling crashes, one notification covers a whole no-progress episode, and any move of that position - drain progress, or the fresh rows of a queue reprovisioned under the same task id, at whatever sequence it restarts - ends that episode and starts a fresh observation interval, while empty, advancing, and declared-wait queues remain silent. Endpointless registered mates remain outside this scan because startup secondmate-liveness owns dead or missing endpoint recovery, and remote homes retain their host-local supervision boundary. -`tests/fm-wake-queue.test.sh` pins the notification, idempotence, quiet-queue, and byte-for-byte foreign-row preservation guarantees. +`tests/fm-wake-queue.test.sh` pins the no-progress notification, drain-progress reset, declared-pause exclusion, active-turn deferral, idempotence, quiet-queue, and byte-for-byte foreign-row preservation guarantees. When a canonical validated PR poll returns exactly `merged`, the watcher routes it through the shared merge-outcome emitter before retiring the poll. [`bin/fm-merge-outcome-lib.sh`](../bin/fm-merge-outcome-lib.sh)'s header owns role routing, PR-specific wake identity, marker-locked normal deduplication, and the at-least-once ordering that prefers a rare duplicate over silence. After successful outcome publication, the watcher immediately delivers the emitter's local actionable poll row and publishes a private retirement receipt bound to the poll's registration, bytes, file identities, metadata, provider, URL, and task ID. @@ -45,7 +52,7 @@ A crew that declares `paused:` for a known external wait, or carries a verified For an ordinary crew that has stopped, the normal-mode watcher first surfaces one stale wake, then applies that same cadence to an unchanged `paused:` or durable `captain-held` endpoint; the pause classification itself is recovered only when the backend confidently reports its agent dead. Live or inconclusive liveness remains fail-open at that initial surface, so a worker genuinely waiting on a decision is never silenced. Its later sights are still held to that same bounded cadence rather than re-alarming on every pane-hash change, because the throttle is keyed to the declaration and not to the pane an idle parked worker keeps ticking. -A secondmate's endpoint liveness is still never read at all; a mate is admitted to that same cadence only to serve a declared wait's bounded re-surface, so a forgotten pause or captain hold on a mate cannot rot invisibly. +A secondmate's endpoint liveness is still never read at all; a mate is admitted to that same cadence only to serve a status-declared wait's bounded re-surface, so a forgotten `paused:` or `captain-held` declaration on a mate cannot rot invisibly. Its initial normal-mode status signal still surfaces through the no-verb path, while away mode self-handles that routine signal and owns the later recheck. Fresh stale panes use the same current-state read before trusting the status log, so an active run or a proven busy worker outranks an old captain-relevant status-log line left behind before validation. No-change heartbeats are also benign. @@ -116,9 +123,9 @@ It suppresses failed-looking closes when the same identity-matched watcher is he Cursor's `bin/fm-turnend-guard-cursor.sh` hook is the same between-turns shape in one synchronous step: it parks the awaited `stop` hook on the arm wrapper and translates an actionable close into one `followup_message`, with a generation baton that makes an older park still running after the next `stop` claim stand down instead of leaking a stale duplicate wake. The existing turn-end guard remains the final backstop for every harness-engine protocol, with pi-signed sharing Pi's protocol, omp's blocking `session_stop` hook compelling one continuation per turn, the `--claude` mode cooperating with the auto-arm claim, and Cursor's `--cursor` mode rendering a block as one bounded follow-up because its `stop` step cannot be blocked. Its `--restart` mode signals only the watcher recorded in the current home's `state/.watch.lock`, so restarting one home cannot kill sibling secondmate watchers. -A pull-based guard (`bin/fm-guard.sh`) warns through supervision tool output if the primary checkout is tangled or if work, process-event sources, registered custom checks, or Relay polling has an unhealthy model-aware supervision verdict; on main it also warns when queued wakes are waiting to be drained. +A pull-based guard (`bin/fm-guard.sh`) warns through supervision tool output if the primary checkout is tangled or if work, process-event sources, registered custom checks, or Relay polling has an unhealthy model-aware supervision verdict; on main it also warns when queued wakes are waiting for main itself to drain. The drain script calls that guard after presenting the queue; records remain durable until the exact generation-bound acknowledgement printed by the drain succeeds after handling, and main may keep the queued-wakes warning visible until then. -The Pi supervision branch's deliberate queued-wake warning exception is owned by [`pi-supervision-branch.md`](pi-supervision-branch.md#components-and-their-owners). +The Pi supervision branch's deliberate queued-wake warning exception is owned by [`pi-supervision-branch.md`](pi-supervision-branch.md#components-and-their-owners), while [`watcher-continuity.md`](watcher-continuity.md#per-actor-acknowledgement) owns the guard's per-actor counting, the advisory main gets for rows a live branch grant holds, and main's retirement of queue rows no actor could ever present or acknowledge. It leads with a prominent bordered tangle banner, while `bin/fm-guard.sh` owns the watcher-down banner and reminder policy so repeated guarded commands stay noisy without reprinting the full banner in the same episode. On every verified primary harness, tracked hook integration gives the primary session a push-based backstop: when work, a process-event source, a registered custom check, or Relay polling needs supervision and no supervision owner provably holds this home with a fresh beacon, blocking-capable Stop hooks block and nonblocking turn-end integrations force one bounded follow-up. The guard covers the main primary and genuinely marked secondmate homes, exempts child crewmate/scout worktrees, is loop-safe per harness, and is documented in [turnend-guard.md](turnend-guard.md). diff --git a/docs/calm-mode-feasibility.md b/docs/calm-mode-feasibility.md index 989e56254ce..288803e8f00 100644 --- a/docs/calm-mode-feasibility.md +++ b/docs/calm-mode-feasibility.md @@ -17,6 +17,7 @@ Pi 0.81.1 was installed when Calm was first built, and Pi 0.82.0 was the later r The inspected Pi CHANGELOG shows no relevant presentation API introduced at either version, so those versions remain verification evidence rather than compatibility bounds. The exported classes used by the adapters (`AssistantMessageComponent` and `InteractiveMode`) are undocumented internals with no stated version guarantee. `tests/fm-calm-pi-extension.test.sh` records the installed Pi version as evidence without gating on it and covers both newer synthetic versions and an unavailable adapter seam. +This host tracks Pi latest, so the version the evidence is pinned to moves; the [2026-09-07 record](#2026-09-07-pi-0851-renderer-and-export-dom-verification) owns the currently pinned version and the renderer comparison behind it. ### Built-in tool override constraints @@ -237,12 +238,12 @@ The test fixture enumerates every class below through the centralized policy, an | `system-notice` | `showStatus`, `showError`, compaction, retry, and startup warning rows | Unsupported boundary; remains visible. | | `cache-notice` | Non-persisted cache-miss `Text` row | Unsupported boundary; remains visible. | | `project-trust-warning` | Non-persisted startup `Text` row | Unsupported boundary; remains visible. | -| `synthetic-user` | Firstmate extension `sendUserMessage`, terminal-injected input, Firstmate-generated Pi positional brief, or the already non-displayed session-start nudge | Canonically classified text-only operational user messages stay ordinary semantic user messages but render through the zero-height adapter (verified on Pi 0.81.1 through 0.82.0) under Calm; legacy entries stay gaplessly controllable, and the session-start nudge retains its existing non-displayed custom-message path. | +| `synthetic-user` | Firstmate extension `sendUserMessage`, terminal-injected input, Firstmate-generated Pi positional brief, or the already non-displayed session-start nudge | Canonically classified text-only operational user messages stay ordinary semantic user messages but render through the zero-height adapter under Calm; legacy entries stay gaplessly controllable, and the session-start nudge retains its existing non-displayed custom-message path. | | `synthetic-assistant` | No authoritative Firstmate source found | Policy-hidden, but Pi exposes no generic assistant-role renderer. | | `unknown` | Future or unclassified transcript component | Policy-hidden, but no generic renderer exists; never claimed as covered. | The installed extension API has no supported global transcript filter, user-message renderer, assistant-message renderer, chat-container API, or generic custom-tool wrapper. -Pi 0.81.1 through 0.82.0 and Pi 0.84.4 export `AssistantMessageComponent` and `InteractiveMode`, so Calm uses separate idempotent, API-probed adapters for assistant thinking layout and the complete operational-user transcript row while leaving all message data and non-Calm rendering unchanged; see the [compatibility contract](calm.md#pi-compatibility) for how a future Pi lacking one of those exports is handled. +Pi 0.81.1 through 0.82.0, Pi 0.84.4, and Pi 0.85.1 export `AssistantMessageComponent` and `InteractiveMode`, so Calm uses separate idempotent, API-probed adapters for assistant thinking layout and the complete operational-user transcript row while leaving all message data and non-Calm rendering unchanged; see the [compatibility contract](calm.md#pi-compatibility) for how a future Pi lacking one of those exports is handled. General component replacement, ANSI cursor erasure, provider-context mutation, and installed-file patching remain rejected as unsupported or preservation-breaking workarounds. ## Cross-harness verification record @@ -286,7 +287,7 @@ The operational provider path covers Calm loaded on, loaded off, default prefere It asserts one persisted and rendered captain answer, exact user-role operational envelopes in order, no replacement custom messages, one processing result, zero operational transcript rows, and the two-row neighboring-assistant geometry for live, adjacent, and restart paths. Quoted current markers, ASCII-only labels, ordinary text before a marker, unrelated U+2063 placement, and image-bearing input remain visible in component and native transcript checks. `tests/fm-pi-primary-live-e2e.test.sh` also proves the working ship replaces the built-in `Working...` row while Calm is active on the credentialed provider path, and that it clears when the run settles, before continuing its ordinary watcher lifecycle. -`tests/fm-pi-primary-types.test.sh` performs strict no-emit TypeScript checking against the installed Pi declarations, currently package version 0.84.4. +`tests/fm-pi-primary-types.test.sh` performs strict no-emit TypeScript checking against whichever Pi declarations are installed, without pinning a version of its own. The relevant commands are: @@ -540,3 +541,71 @@ FM_TEST_END 2026-08-29T01:01:30Z tests/fm-pi-branch-extension.test.sh exit=0 dur ``` The real renderer comparison exercised twelve outcome lines and reported collapsed and expanded parity with Pi stock, zero visible rows under Calm, restored stock parity after toggling Calm off, and delegated stock HTML export fallback. + +## 2026-09-07 Pi 0.85.1 renderer and export-DOM verification + +This host tracks Pi latest, so the version this contract's evidence is pinned to moves. +The renderer and lifecycle evidence below was taken against installed `@earendil-works/pi-coding-agent` 0.85.1 with `@earendil-works/pi-server` 0.85.0 also installed globally. + +Calm's rendered rows are unchanged across 0.84.4, 0.85.0, and 0.85.1. +`FM_PI_PACKAGE_DIR` points `tests/fm-calm-pi-extension.test.sh` at an isolated install, so each comparison ran against its own temporary dependency tree and never mutated the globally installed packages. + +```text +$ pi --version +0.85.1 + +$ npm ls -g --depth 0 @earendil-works/pi-coding-agent @earendil-works/pi-server +├── @earendil-works/pi-coding-agent@0.85.1 +└── @earendil-works/pi-server@0.85.0 +``` + +```text +$ FM_PI_PACKAGE_DIR= tests/fm-calm-pi-extension.test.sh +ok - Pi calm centralizes transcript visibility, preserves execution/export data, keeps Pi's stock working row visible while no run is active, and persists its choice across session starts +$ FM_PI_PACKAGE_DIR= tests/fm-calm-pi-extension.test.sh +ok - Pi calm centralizes transcript visibility, preserves execution/export data, keeps Pi's stock working row visible while no run is active, and persists its choice across session starts +$ FM_PI_PACKAGE_DIR= tests/fm-calm-pi-extension.test.sh +ok - Pi calm centralizes transcript visibility, preserves execution/export data, keeps Pi's stock working row visible while no run is active, and persists its choice across session starts +``` + +Reaching that parity on 0.85 took one contract adaptation, landed earlier in 85ad5e7. +Pi 0.84 and older silently substituted a built-in's stock definition when a `ToolExecutionComponent` was constructed without one, so the calm-off equivalence baseline could be built definition-less and still read as stock. +Pi 0.85 removed that substitution, so the definition-less baseline renders Pi's generic text fallback instead - which is what produced `read collapsed rendering changed while calm mode was off`. +The renderer change was real, and it was the contract's baseline that had to adapt, not Calm's wrappers: the wrapped rows matched Pi stock before and after. +`tests/fm-calm-pi-extension.test.sh` now builds each baseline from the real stock tool-definition factories that `dist/core/tools/index.js` exports, calling the built-in's own factory with `process.cwd()`, which reads as stock on 0.84.4 and on 0.85.x alike and no longer depends on the removed substitution. + +Pi 0.85.0 alone requires a package it does not declare. +Its `dist/experimental/server.js` statically imports `@earendil-works/pi-server`, which is absent from 0.85.0's `dependencies`, `peerDependencies`, and `optionalDependencies`, so a clean install of 0.85.0 on its own cannot load Pi's interactive mode at all: + +```text +Error [ERR_MODULE_NOT_FOUND]: Cannot find package '@earendil-works/pi-server' imported from + .../node_modules/@earendil-works/pi-coding-agent/dist/experimental/server.js +``` + +Installing `@earendil-works/pi-server@0.85.0` beside it restores the identical Calm rendering, and 0.85.1 no longer reaches that import. +That packaging gap is a separate installation defect, not the renderer change above: it stops Pi from loading at all rather than altering any rendered row. + +The `could not render calm-mode HTML export DOM` failure was a headless-Chrome start-up flake, not a change in Pi's export shape. +It appeared in exactly one of the thirteen most recent CI runs, and that run installed the same Pi 0.85.1 as the runs immediately before and after it, which both passed. +The render step is a vendor-tool step: the assertions that follow it are what protect the Calm conversation boundary. +It now retries a bounded number of Chrome start-ups on a fresh profile and, when every attempt fails, reports the Chrome binary, its version, the installed Pi version, each attempt's exit status, whether that attempt was timed out, and Chrome's own stderr, so the next occurrence is diagnosable from the CI log alone. +`test_export_dom_render_guard` in the same script pins that behavior with real processes and no browser. + +The complete Calm suite against installed Pi 0.85.1, with `FM_CHROME_BIN` naming the Chrome the render step used: + +```text +$ FM_CHROME_BIN= tests/fm-calm-pi-extension.test.sh +ok - Pi calm resolves its persistent home independently of Pi's launch directory +ok - Pi calm compatibility evidence never rejects a Pi version for being newer than 0.82.0, and still fails closed on a missing or malformed version +ok - a missing collapsed-thinking presentation API degrades only that Calm adapter with a clear skip reason, while the rest of Calm still registers +ok - missing Pi presentation class exports reach the independent adapter degradation path +ok - Calm registers none of its 7 built-in tool wrappers at load while config/calm is off, and all 7 synchronously at load while config/calm is on +ok - Calm's first same-session /calm activation claims every uncontested built-in, leaves a foreign bash tool fully intact and callable, warns prominently and logs the contested name, and only rows constructed before that activation - the documented bound - fail to retroactively collapse +ok - Pi calm centralizes transcript visibility, preserves execution/export data, keeps Pi's stock working row visible while no run is active, and persists its choice across session starts +ok - Pi calm on collapses mid-turn assistant working notes to zero height while Calm off keeps them, leaves streaming, truncated-final, and genuine final replies untouched, never mutates the messages, ignores every /calm argument, and restores a legacy persisted max as ordinary Calm on +ok - Pi operational follow-up E2E processes exact user-role notifications once while Calm hides current and adjacent rows, Calm off and absent render them, and restart preserves semantics +ok - Pi Calm native /skill:ahoy geometry keeps every collapsed thinking and tool block at zero height while preserving expansion, history, restart, and Calm-off rendering +ok - Pi Calm working ship moves on a slow independent cadence over faster fixed-cell blue water, paints the complete boat standard yellow with balanced resets, keeps ANSI-stripped width exact, flips the directional sail on the exact bounce at both edges and every width, clamps visible and hidden resizes, falls back deterministically when narrow, freezes and resumes column/direction across settle/start without hidden-time jumps or duplicate timers, resets only on a fresh session, and installs and removes one scheduler-owning widget across starts, settle, abort, failure, shutdown, reload, replacement, and Calm toggles while leaving Calm-off visibility untouched +ok - the rendered-export-DOM guard renders in one pass, retries a bounded number of Chrome start-up failures, and reports the Chrome binary, Chrome version, Pi version, exit status, and Chrome diagnostic when every attempt fails +ok - Pi calm native E2E replaces the stock working row with a moving, resize-clamped working ship that freezes and resumes across two working periods in one Pi session, clears on abort, keeps captain turns visible, hides exact operational user rows without changing persistence, restores stock rendering Calm-off, survives restart, and preserves export plus Ctrl+O behavior +``` diff --git a/docs/captain-hold-lifecycle.md b/docs/captain-hold-lifecycle.md index 41ee23d2243..842846ea1f3 100644 --- a/docs/captain-hold-lifecycle.md +++ b/docs/captain-hold-lifecycle.md @@ -186,7 +186,7 @@ It also proves the two verification outcomes - an evidence-backed `reconciled` c The captured-source coverage proves Lavish deduplicates each card before separating versioned structured selections from notes, bare and annotated Reconcile choices never reach keyed answers, genuine current and legacy choices still close normally, legacy bare and separator-annotated reconcile values feed neither intake, mixed repeated selections preserve every other card's final value, the generic runner creates a request only through a verified bound source, chat reconcile text creates none, and the resulting board request authorizes evidence-backed closure. The board's half is pinned in `tests/fm-bearings-board.test.sh`: every published decision card carries exactly one reconcile option, authored options reserve that value across every card type, recommendations name authored options, a decision card whose structured subject appears in the payload's landed rows is dropped while a genuinely open one is kept even when an unrelated landed id contains its key after a newline, a build requires a fresh authoritative listed-open result before binding or arming, a reopen retires the pre-reopen source generation and waits for a fresh live listener, and a rebuild of an already-armed board with no live listener starts one. That suite drives its Lavish session through a protocol-shaped stub, and `tests/fm-bearings-board-lavish-live-e2e.test.sh` is the default-on capability guard for the installed provider; [`verification/process-event-sources.md`](verification/process-event-sources.md) owns the version-scoped evidence. -`tests/fm-procevent.test.sh` pins the ownership half: a dead generation whose recorded state-root identity no longer matches is reclaimed by reconcile into a replacement that actually runs, `retire` releases the same claim instead of refusing, and neither a live generation nor a crashed leader whose owned group survives is reclaimed under that same drift. +[`verification/process-event-sources.md`](verification/process-event-sources.md) owns the process-event ownership and reclamation evidence exercised by `tests/fm-procevent.test.sh`. `tests/fm-classify-decision-key.test.sh` pins `status_key_closing_verb` itself: it separates a resolution from the durable-transfer close and from a still-open key, reports the last real transition across re-openings and both key positions, and treats a prose mention as no transition. diff --git a/docs/configuration.md b/docs/configuration.md index 2ee007121d2..4fd711f011f 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -59,7 +59,8 @@ The effort list is a handful of levels and stays on Pi's plain selector dialog. Both picks change the supervision branch alone and never the captain's own conversation model or effort. It persists the model pick in gitignored `config/supervision-branch-model` and the effort pick in gitignored `config/supervision-branch-effort`, both under the effective Firstmate home, resolved from `FM_HOME`, then `FM_ROOT_OVERRIDE`, then the tracked code root derived from the extension path, or under `FM_CONFIG_OVERRIDE` when that test and specialized-setup override is present. Firstmate keeps no model catalog of its own; the list is the intersection of what Pi reports when the picker opens and what a fresh isolated branch runtime can run. -A provider that exists only because an extension registered it inside the captain's session is not offered, while stored OAuth and API-key credentials retain their native credential type because Firstmate never copies, converts, installs, or overwrites credentials for the branch runtime. +A provider that exists only because an extension registered it inside the captain's session, such as pi-devin-auth's `devin`, is offered and can be pinned or followed like any other; [pi-supervision-branch.md](pi-supervision-branch.md#cost-model-and-the-byte-stable-prefix) owns how that registration reaches the isolated branch runtime. +Stored OAuth and API-key credentials retain their native credential type because Firstmate never copies, converts, installs, or overwrites credentials for the branch runtime. The file holds one `/` line followed by one newline, split at the first `/` so a provider-qualified model id such as `openrouter/anthropic/claude-sonnet-4-5` survives intact. An absent, unreadable, or unparseable file means no pin, and the branch then follows main's own current model, applied explicitly and live whenever main changes models mid-session. A valid pin wins over main and remains unaffected by main's model changes. @@ -384,6 +385,8 @@ The filter runs at the worker command boundary, after the terminal daemon and pa This is not a sandbox: it cannot revoke same-user access to credential files, prevent tools or later shells from loading credentials again, or isolate processes from the same user's other processes. Regression coverage executes emitted launch commands with synthetic nonsecret values in [`tests/fm-spawn-dispatch-profile.test.sh`](../tests/fm-spawn-dispatch-profile.test.sh). +Every claude launch's inline `--settings` JSON also carries `"attribution":{"commit":"","pr":"","sessionUrl":false}`, so a spawned worker never writes a Co-Authored-By trailer, Claude-Session link, or generated-with line into a commit or PR body regardless of which settings scopes end up loaded. + ## Crew dispatch profiles (config/crew-dispatch.json) `config/crew-dispatch.json` is an optional local, gitignored file containing natural-language rules that firstmate reads before dispatching a crewmate or scout. @@ -728,8 +731,9 @@ Never run the registered blocking source command directly in a conversational tu A long-polling external process is registered as a *source* through its adapter, whose header and `--help` own the commands and flags. `bin/fm-procevent.sh` owns the generic contract; built-in adapters retain their tracked `bin/fm-procevent-.sh` commands, while an explicitly bound external adapter routes through the trusted host contract above. `bin/fm-procevent-lavish.sh` is the first built-in adapter and wraps only the currently published `lavish-axi poll` interface. -That adapter, and only that adapter, retries the one exact transient response a cut-short listener returns while its marks remain available (`error: Lavish Editor poll response was interrupted` with `code: SERVER_ERROR`), up to 12 times at 5 second intervals, so an internal retry never reaches the runner as a captured result. -Real feedback, ended and missing sessions, any other `SERVER_ERROR`, and that same interruption still standing once the bound is spent are all captured and announced normally; `FM_LAVISH_POLL_RETRY_DELAY` is a bounded 0 to 60 second test override for the interval only, and the runner itself stays adapter-agnostic. +That adapter, and only that adapter, retries the one exact transient response a cut-short listener returns while its marks remain available (`error: Lavish Editor poll response was interrupted` with `code: SERVER_ERROR`), up to 12 times with poll starts at least 5 seconds apart, so an internal retry never reaches the runner as a captured result. +This start-to-start governor is a no-op after a normally blocking poll but caps an immediately returning poll under the shipped defaults independently of the owner lease and registration launch pacing. +Real feedback, ended and missing sessions, any other `SERVER_ERROR`, and that same interruption still standing once the bound is spent are all captured and announced normally; `FM_LAVISH_POLL_RETRY_DELAY` is a bounded 1 to 60 second test override for the interval only, and the runner itself stays adapter-agnostic. An already-armed Lavish source keeps its registered listener command until it is retired and armed again, so re-arm a live board once to adopt this retry policy. The `when` adapter (`bin/fm-procevent-when.sh`) turns this channel into a condition->action primitive: it registers a deterministic condition and a deterministic action once, its blocking child polls the condition without waking firstmate, and a stable true fires the action at most once before one terminal outcome is durably captured and published as a wake that remains eligible for re-announcement until handled. @@ -787,14 +791,16 @@ Claims live under `$XDG_STATE_HOME/firstmate/procevent-claims` (override with `F Each claim binds its caller-reported home and runner PID to a process identity, unique claim generation, exact registration-file generation, and resolved state-root identity. Registration, acquisition, replacement, retirement, and generation-bound release are serialized at one machine-wide boundary per source. A live identity-matched owner is never displaced, and release removes only the exact generation the caller acquired. -Retirement and orphan reconciliation signal a runner process group only while its recorded process identity still matches, or when the recorded leader is gone and only its own owned group survives. -A runner leads its own process group, so a claim counts as reclaimable only when its owner is stale and an independent process-group check finds no members; a crashed leader or reused pid whose old group still has members cannot relax ownership cleanup, and reconcile stops a safely identified surviving group before starting any replacement. +Retirement and orphan reconciliation select a runner process group for signalling only while its recorded process identity still matches and the live runner still leads that group. +A claim counts as reclaimable only when its owner is stale and an independent process-group check finds no members; a crashed leader or reused pid whose process group still has members cannot relax ownership cleanup, so reconcile preserves the claim without signalling the ambiguous group or starting a replacement. Reclaiming a generation that IS gone is not gated on tidying its capture-reservation records. Those records are keyed by claim token and every replacement claims a fresh one, so a leftover that can no longer be located - a state-root identity a claim recorded before its home was re-created, for example - is stale bytes rather than an ownership hazard. Ordinary release and reclamation still attempt reservation cleanup and require it unless both owner staleness and whole-group absence prove the generation gone. The narrow live-owner terminal-self-retirement path also attempts cleanup but tolerates its own still-in-flight reservation, which the runner removes on the normal end-of-capture path; exact home, PID, and claim-token ownership remains mandatory before the claim is released. If identity cannot be established for a live PID, or a surviving owned group cannot be proved stopped, the operation preserves the registration and claim for safe retry rather than adding a second owner. -A live PID whose identity no longer matches is a reused PID, so it is treated as stale and its process group is never signalled. +A live PID whose identity no longer matches is a reused PID, so cleanup refuses it before signalling. +Identity and process-group verification cannot be made atomic with signalling in portable shell: the reaper signals only a target it has verified as the recorded generation, but PID and group reuse remain possible in the narrow interval between verification and the signal. +Launch pacing is the primary host-wedge protection; watchdog cleanup is a backstop. Supported secondmate retirement preflights each target home's bounded `sweep-home` command before destructive teardown, snapshots its registrations outside the target, then runs the sweep at that home's final deletion or return boundary. If deletion or return fails, teardown restores those registrations and reconciles them before returning the refusal. @@ -803,6 +809,26 @@ The sweep retires local registrations and machine-wide claims whose recorded sta Teardown refuses with the home, lease, routing evidence, registrations, claims, and runners retained when identity is uncertain, ownership is unreadable or unreleased, or relevant state exists without a sweep-capable child script. Raw manual deletion of a Firstmate home is unsupported because it can orphan a blocking child. To recover, restore that home's tracked `bin/fm-procevent.sh`, run `FM_HOME= /bin/fm-procevent.sh sweep-home`, then rerun the supported teardown. +The owning-home lease below bounds how long such an orphan can run, but it is a backstop, not a substitute for the supported path. + +A runner is bound to the HOME that owns it, not to the one session that armed it. +That granularity is deliberate: a persistent source is meant to outlive the turn and the session that armed it, so binding a runner to its arming session would stop exactly the sources this mechanism exists to keep running. +Any activity in the same home refreshes the lease, so a replacement session, another watcher, or an ordinary inspection command keeps a runner of that home alive; a runner whose SOURCE is no longer wanted in a live home is stopped by reconcile when that source is retired, independently of the lease. +The lease is therefore the backstop for a home that is GONE - the torn-down test sandbox this change exists to bound - and not a per-session ownership check. +KNOWN LIMIT: while any activity continues in a home whose original owning session has ended, that activity refreshes the lease and a runner of that home keeps running until its source is retired or the home goes away. +Detaching a runner into its own process group is what lets a persistent source outlive the turn that armed it, and on its own it is also what lets a runner outlive its whole home: reparented to init, it keeps its blocking child - and every process that child spawns - running with nothing left to reap it. +So a home's process-event state carries a lease that registration, attached start, reconciliation, acknowledgement, and listing refresh, and the watcher's reconcile cycle is what keeps it fresh in a live home. +An attached public `start` continues refreshing the lease while its caller remains attached. +Each runner fails closed unless a small guard starts successfully beside it in a separate process group. +That guard accepts the lease only while the state root retains the device/inode identity recorded by the runner's claim, and stops the runner's whole process group after two consecutive checks cannot prove that identity and lease freshness. +The group signal reaches the blocking child and everything under it exactly as retirement does. +A runner exports the inherited `FM_PROCEVENT_IN_RUNNER` marker and every lease refresh is skipped under it, so a runner and its ordinary children do not certify their own owner, and the next reconcile in a live home simply starts a replacement runner. +That no-self-refresh rule is CONFUSED-AGENT-GRADE, the same deliberate captain-decided grade `bin/fm-lease-lib.sh` documents: it stops the accidental case this boundary exists for, an orphaned or test-scaffolding source tree that would otherwise keep its own owner alive. +A source that DELIBERATELY strips the marker from its environment can still refresh the lease, so adversarial-grade unforgeability is explicitly out of scope here and tracked as separate follow-up design work. +Scope is the owning state root and one runner generation, never a script or process name, so a live source in another home is untouched: that home refreshes its own lease. +`FM_PROCEVENT_OWNER_LEASE_SECONDS` (default 600, range 1..86400) is how long a runner keeps going with no sign of activity in its owning home, and `FM_PROCEVENT_OWNER_CHECK_SECONDS` (default 15, range 1..3600) is how often its guard re-reads the lease. +`FM_PROCEVENT_LAUNCH_FLOOR_SECONDS` (default 1, range 1..3600) is the minimum time between consecutive launches of one registration generation's stored command, bounding the launch rate of an immediately returning source during that lease window. +The generation's first launch is immediate, later launches share its monotonic pacing timestamp, a timestamp from before a reboot is treated as expired, and replacing the registration starts a fresh pacing generation. `FM_PROCEVENT_MAX_OUTPUT_BYTES` (default 1048576) bounds a single captured result while the source runs; oversized output is drained but truncated with a stderr notice rather than staged or published whole or dropped. @@ -891,6 +917,9 @@ FM_TOOL_UPDATE_BUDGET_SECS=20 # 1..120 seconds allowed for a whole watched-too FM_TOOL_UPDATE_NOW= # test override for the watched-tool sweep clock; the sweep budget still uses real time FM_PROCEVENT_MAX_OUTPUT_BYTES=1048576 # bound on one captured process-to-event result FM_PROCEVENT_CLAIM_ROOT= # machine-wide source claim root; default $XDG_STATE_HOME/firstmate/procevent-claims +FM_PROCEVENT_OWNER_LEASE_SECONDS=600 # how long a source runner keeps going with no activity in its owning home; 1..86400 +FM_PROCEVENT_OWNER_CHECK_SECONDS=15 # how often a runner's guard re-reads that lease; 1..3600 +FM_PROCEVENT_LAUNCH_FLOOR_SECONDS=1 # minimum interval between launches of one registration generation's source command; 1..3600 FM_WHEN_OUTPUT_TAIL_BYTES=8192 # bound on the command-output tail inside one condition->action outcome document FM_CODEX_WATCH_CHECKPOINT=180 # seconds per foreground watcher checkpoint in Codex primary supervision FM_CREW_STATE_NM_TIMEOUT=10 # seconds allowed per no-mistakes query inside fm-crew-state.sh @@ -924,15 +953,15 @@ FM_WATCH_REARM_RETRY_MAX_MS=4000 # Pi/OpenCode adapter cap for exponential con FM_WATCH_REARM_RETRY_LIMIT=5 # Pi/OpenCode adapter launch-failure retries before surfacing restoration failure FM_WATCH_CYCLE_LOG_MAX_BYTES=262144 # size cap for the arm-owned watcher lifecycle ledger FM_WATCH_CYCLE_LOG_KEEP_LINES=1000 # newest complete lifecycle rows considered when the ledger is capped -FM_WATCHER_STALE_GRACE=300 # defaults to FM_GUARD_GRACE; seconds a live watcher lock may have a stale beacon before re-arm errors +FM_WATCHER_STALE_GRACE=300 # defaults to FM_GUARD_GRACE if set, else the poll-derived grace (docs/turnend-guard.md "Guard grace and the poll cadence"); seconds a live watcher lock may have a stale beacon before re-arm errors FM_SIGNAL_GRACE=30 # seconds to coalesce nearby status and turn-end signals into one wake FM_TURNEND_CHURN_ABSORB_SECS=900 # longest one endpoint's bare turn-ends may be deferred on pane-churn evidence alone; only consulted when config/turnend-churn-absorb is present FM_CAPTAIN_RE='done:|needs-decision:|blocked:|failed:|PR ready|checks green|ready in branch|merged' # captain-relevant status regex; nonterminal progress verbs remain excluded even when their prose matches FM_CLASSIFY_PAUSED_VERB=paused # leading status verb for a declared external wait; excluded from FM_CAPTAIN_RE and distinct from blocked FM_STALE_ESCALATE_SECS=240 # idle seconds before a provably-working stale pane escalates; stale panes whose crew is not provably working surface immediately unless admitted directly to the declared-wait cadence, while a live idle declared wait still surfaces once before that cadence bounds repeats FM_BUSY_TURN_MAX_SECS=3600 # maximum age of a busy pane's latest state/.turn-ended marker, or its state/.meta spawn record before any turn completes, before the same wedge escalation used for a provably-working non-busy stale takes over; inspection-only, never an automatic interrupt or restart; a declared external wait or verified captain-held transfer takes the FM_PAUSE_RESURFACE_SECS recheck below instead -FM_PAUSE_RESURFACE_SECS=3600 # seconds between bounded rechecks of a declared external wait or verified captain-held transfer, including a live idle pane after its first inconclusive stale wake and a live busy pane past FM_BUSY_TURN_MAX_SECS; the away-mode daemon uses the same setting, ageing its window against the crew's own latest status line rather than pane busy state -FM_SECONDMATE_WAKE_STALL_SECS=60 # minimum age of the oldest valid foreign wake-queue row before an endpoint-recorded local secondmate produces one durable parent wake-loop-stall notification; zero or invalid values use 60 +FM_PAUSE_RESURFACE_SECS=3600 # seconds between bounded rechecks of a declared external wait or verified captain-held transfer, and between repeated new-hash stale alarms for an ordinary crew task with an open backlog captain call; this includes a live idle pane after its first inconclusive stale wake and a live busy pane past FM_BUSY_TURN_MAX_SECS, while the away-mode daemon uses the same setting and ages its window against the crew's own latest status line rather than pane busy state +FM_SECONDMATE_WAKE_STALL_SECS=180 # minimum interval with no change of the oldest actionable foreign wake-queue row (it advances as the mate drains, and a queue reprovisioned under the same task id starts a fresh interval at whatever sequence it restarts) before an endpoint-recorded local secondmate produces one durable parent wake-loop-stall notification for that no-progress episode; a mate that is provably inside an active turn (an exact busy verdict, bounded by the same FM_BUSY_TURN_MAX_SECS above) never escalates whatever this interval says, declared external-wait pause rows are excluded, and zero or invalid values use 180 FM_WEDGE_DEMAND_INSPECT_COUNT=3 # consecutive provably-working stale escalations on the same unchanged pane before demand-deep-inspection is added FM_WORKTREE_WRITE_PRUNE='.git node_modules .venv venv __pycache__ .mypy_cache .pytest_cache .ruff_cache .tox target dist build .next .cache vendor' # directory names the wedge detector's task-worktree write probe skips; the default keeps .git out so a supervisor's own read-only git command can never look like crew progress; set it to the empty string to prune nothing, which widens the probe to the whole depth-bounded tree rather than disabling it FM_WORKTREE_WRITE_MAXDEPTH=6 # depth that same probe walks below the recorded worktree; it runs only at the moment a wedge escalation would otherwise fire, never on every poll; no probe knob applies to a secondmate, whose recorded worktree is a provisioned home the probe skips entirely diff --git a/docs/fm-test-portable-shards.md b/docs/fm-test-portable-shards.md index e1a2dd06655..c966f1b9c27 100644 --- a/docs/fm-test-portable-shards.md +++ b/docs/fm-test-portable-shards.md @@ -64,10 +64,10 @@ Each shard is still strictly serial in itself, and separate runners mean no two `.github/workflows/ci.yml` derives the same `n` from `strategy.job-total` rather than a literal, so changing the shard count in either file without the other fails the lane loudly instead of leaving part of the required suite unrun. Assignment is longest-processing-time bin packing over per-script duration hints embedded in `bin/fm-test-run.sh`. -The 141 current hints include the slowest measurements retained from the `fm-test-timing-portable-serial-*` artifacts of three green CI runs on 2026-09-01, [33558082172](https://github.com/kunchenguid/firstmate/actions/runs/33558082172), [33523597838](https://github.com/kunchenguid/firstmate/actions/runs/33523597838), and [33463326167](https://github.com/kunchenguid/firstmate/actions/runs/33463326167), plus the 5121 ms native-Windows focused runner measurement for `tests/fm-pi-windows-shell-invocation.test.sh` from 2026-09-06T21:02Z. -Those per-script maxima total 3830189 ms of conservative balance weight. +The 142 current hints include the slowest measurements retained from the `fm-test-timing-portable-serial-*` artifacts of three green CI runs on 2026-09-01, [33558082172](https://github.com/kunchenguid/firstmate/actions/runs/33558082172), [33523597838](https://github.com/kunchenguid/firstmate/actions/runs/33523597838), and [33463326167](https://github.com/kunchenguid/firstmate/actions/runs/33463326167), plus the 5121 ms native-Windows focused runner measurement for `tests/fm-pi-windows-shell-invocation.test.sh` from 2026-09-06T21:02Z. +Those per-script maxima total 3836189 ms of conservative balance weight. Taking the slowest of several CI runs rather than a single run keeps the balance honest on a slow runner: individual scripts varied by up to 20% between those three runs. -A script with no hint gets the conservative `PORTABLE_SERIAL_DEFAULT_WEIGHT_MS` default; the current 147-script lane has six such scripts, bringing its assignment weight to 3992189 ms. +A script with no hint gets the conservative `PORTABLE_SERIAL_DEFAULT_WEIGHT_MS` default; the current 152-script lane has ten such scripts, bringing its assignment weight to 4106189 ms. Hints only affect balance: the coverage guard keeps the partition complete and disjoint whatever they say, so a stale hint costs a slower shard rather than lost coverage. Balance is still worth keeping current, because enough unmeasured scripts let one shard carry more than twice another shard's real work and reach the job cap while another runner sits idle. That is not hypothetical: by 2026-09-01 the lane had grown from 116 to 139 scripts and from ~42 to ~63 minutes, 17 scripts were still unmeasured, and several hints were low by 2-5x, so shard 3 of 4 ran 17-20 minutes against its 20-minute cap while shard 1 ran 11.5 minutes and run [33574154856](https://github.com/kunchenguid/firstmate/actions/runs/33574154856) timed out seconds after a passing test. @@ -76,14 +76,14 @@ Refresh the hints whenever the serial lane gains scripts, rather than waiting fo | Lane | Script count | Estimated duration | |---|---:|---:| -| `portable-serial-1of5` | 29 | 798443 ms (~13.31 min) | -| `portable-serial-2of5` | 30 | 798440 ms (~13.31 min) | -| `portable-serial-3of5` | 29 | 798432 ms (~13.31 min) | -| `portable-serial-4of5` | 29 | 798434 ms (~13.31 min) | -| `portable-serial-5of5` | 30 | 798440 ms (~13.31 min) | -| imbalance | | 11 ms | - -The current table is generated from the runner's retained maxima plus its default for the six unhinted scripts. +| `portable-serial-1of5` | 29 | 821231 ms (~13.69 min) | +| `portable-serial-2of5` | 30 | 821243 ms (~13.69 min) | +| `portable-serial-3of5` | 31 | 821236 ms (~13.69 min) | +| `portable-serial-4of5` | 31 | 821247 ms (~13.69 min) | +| `portable-serial-5of5` | 31 | 821232 ms (~13.69 min) | +| imbalance | | 16 ms | + +The current table is generated from the runner's retained maxima plus its default for the ten unhinted scripts. The last complete replay against the three source runs put the then-current partition's worst shard at 12.54 min, 63% of the 20-minute job cap. The single longest script, `tests/fm-watch-triage.test.sh` at 262626 ms, is the floor for any shard count. @@ -126,7 +126,7 @@ Portable shards, each portable serial shard, and the Herdr lane upload runner-ge | Lane | Bound | Rationale | |---|---|---| | portable parallel 1/2 | job `timeout-minutes: 10` | The measured shard sums are about three minutes and the timeout is a hang tripwire. | -| portable serial 1-5 | job `timeout-minutes: 20` | Each balanced shard carries about 13.31 minutes of conservative assignment weight, leaving roughly 1.5x hang-tripwire margin for job setup and runner-speed spread. | +| portable serial 1-5 | job `timeout-minutes: 20` | Each balanced shard carries about 13.69 minutes of conservative assignment weight, leaving roughly 1.5x hang-tripwire margin for job setup and runner-speed spread. | | Herdr | family-run step `timeout-minutes: 20`; job `timeout-minutes: 75` backstop | Healthy runs finished around 7 minutes before this lane gained `fm-backend-herdr-focus-flash-e2e`, which measures about 2 minutes against a real lab locally, so the step bound is still the hang tripwire (cleanup and timing artifacts still upload) while the job cap stays a last-resort backstop. Refresh this figure from the lane's uploaded timing artifact. | Timeouts are hang tripwires rather than expected healthy durations. diff --git a/docs/pi-supervision-branch.md b/docs/pi-supervision-branch.md index 144d0c76f5d..76cb84a8b85 100644 --- a/docs/pi-supervision-branch.md +++ b/docs/pi-supervision-branch.md @@ -144,6 +144,8 @@ Every other fleet-wide or unresolvable wake - including watcher-failure alarms, The captain accepted the normal provider prompt-caching strategy: a byte-identical branch prefix generated once per firstmate version, the same tool set in the same order on every request, and one shared `prompt_cache_key` per home for all branch sessions (set in a `before_provider_request` hook, and only for providers whose requests already carry that field); main keeps its own per-session key. Budget roughly 60% cache hits on a new branch conversation's first call and 95% on later calls within that conversation; the shared per-home key is what carries the byte-identical prefix across the conversation each main session start opens, and reuse is best-effort, never guaranteed. The branch can also run on a cheaper model and a shallower reasoning effort than main, both pinned with the Pi `/supervision-model` command; [configuration.md](configuration.md#pi-supervision-branch-model-and-effort-configsupervision-branch-model-configsupervision-branch-effort) owns those pins' operator-facing schema and unpinned behavior. +A provider an extension registered only into main's runtime, such as pi-devin-auth's `devin`, reaches the isolated branch runtime by copying its provider config from main's captured `ModelRegistry` into the branch `ModelRuntime` at model-resolution time and in the `/supervision-model` picker, so the provider's own `streamSimple` transport and OAuth wiring are reused by reference rather than reimplemented. +That carve-out is scoped to provider registration alone: the branch keeps its `noExtensions`, `noSkills`, and `noContextFiles` isolation, the copy is never persisted, a provider whose registration fails to compose is simply unavailable, and `tests/fm-pi-branch-extension.test.sh` pins the pin-and-fallthrough behavior. No caching machinery beyond this exists, deliberately: any later dynamic content in the branch prefix silently removes most of the cache benefit, which is why `bin/fm-branch-prompt.sh`'s header is the contract's single owner and `tests/fm-branch-supervision.test.sh` pins the output to byte identity. ## Away mode diff --git a/docs/turnend-guard.md b/docs/turnend-guard.md index d33b36f0f55..a3e73dfb54f 100644 --- a/docs/turnend-guard.md +++ b/docs/turnend-guard.md @@ -49,13 +49,24 @@ While `state/.afk` exists the away-mode daemon (`bin/fm-supervise-daemon.sh`) ow The turn-end guard therefore accepts `fm_afk_daemon_owns_supervision` from `bin/fm-wake-lib.sh` as proof of supervision on that path: away mode must be active, and this home's `state/.supervise-daemon.lock` must name a live pid whose current process identity still matches the identity the daemon recorded for itself. That is the same identity discipline the watcher lock uses, so a recycled pid, a lock left behind by a killed daemon, and a daemon that never recorded its identity all fail it. A daemon that cannot record its own identity at startup logs a warning and keeps running, because a supervisor must not refuse to run over an unreadable `ps`; that warning is what names the cause when the guard then keeps blocking away-mode turn boundaries for the rest of that daemon's life. -The proof covers ownership only, never freshness: the fresh-beacon half of the predicate is unchanged, so a daemon that stops restarting its watcher still blocks once the beacon passes grace, and a home with no daemon and no watcher blocks exactly as it did before. +The proof covers ownership only, never freshness: the guard still requires a fresh beacon, so a daemon that stops restarting its watcher still blocks once the beacon passes grace, and a home with no daemon and no watcher blocks exactly as it did before. +That beacon check uses the poll-derived grace described below rather than the flat `FM_GUARD_GRACE` default, because the daemon starts a fresh one-shot watcher only after it finishes handling the previous wake, and that handling can legitimately outrun a fixed 300-second window under load (a slow registered check, a busy supervisor pane) with the daemon perfectly healthy throughout. With away mode off the daemon lock proves nothing and the strict watcher predicate is unchanged. `FM_STATE_OVERRIDE` wins over `FM_HOME/state`, and `FM_HOME` wins over repository-root `state/`. `FM_GUARD_GRACE` controls beacon freshness and defaults to 300 seconds. If `jq` is missing or hook stdin is empty, the guard exits 0 because it cannot safely read loop-guard fields. +### Guard grace and the poll cadence + +`bin/fm-watch.sh` touches `state/.last-watcher-beat` once per cycle, immediately before its terminal wait (`event_wait_or_sleep`) as well as at the top of the next cycle, so a healthy watcher's beacon can legitimately age up to `FM_POLL` seconds between touches. +A fixed 300-second grace default stops correctly bounding staleness once a home's `FM_POLL` reaches or exceeds it: a perfectly healthy watcher mid-wait would then read stale at the edge of every full poll cycle by definition, which is exactly what a long-poll home (`FM_POLL=300`) hit against the Claude Stop-hook auto-arm (`bin/fm-claude-stop-autoarm.sh`). +That hook and `bin/fm-watch.sh`'s own pre-acquisition staleness check (the "lock held by live pid but heartbeat is stale" refusal) both derive their default grace from the configured poll instead of a bare constant: `max(300, FM_POLL + 60)`, so the default never drops below the historical 300-second floor for the common short-poll case but grows with the poll cadence once that cadence would otherwise outrun it. +`fm_poll_derived_grace` in `bin/fm-wake-lib.sh` is the single owner of that formula. +The auto-arm hook additionally exports its resolved `FM_GUARD_GRACE` when it forks `bin/fm-watch-arm.sh`, so the arm wrapper and the watcher it may start judge staleness with the exact same value the hook just judged it with, whether that value came from an operator override or the poll-derived default. +`bin/fm-turnend-guard.sh`'s away-mode branch (`fm_afk_daemon_owns_supervision`, above) also derives its beacon grace from `fm_poll_derived_grace` rather than falling back to the bare 300-second default, for the same reason: the daemon's watcher-restart cadence there is not a fixed poll loop, so a flat grace misreads a daemon that is genuinely still cycling as down. +Every other direct `FM_GUARD_GRACE` reader (`bin/fm-guard.sh`, the strict-watcher checks in `bin/fm-turnend-guard.sh` and its harness-specific wrappers, `bin/fm-wake-lib.sh`) still falls back to the bare 300-second default unless `FM_GUARD_GRACE` is set explicitly in the environment. + ## Harness integrations - Claude registers two `Stop` hooks in `.claude/settings.json`, both anchored through `CLAUDE_PROJECT_DIR`: `bin/fm-turnend-guard.sh --claude`, and `bin/fm-claude-stop-autoarm.sh` with `asyncRewake: true` and `timeout: 28800`. @@ -170,7 +181,7 @@ That warning uses `bin/fm-supervision-instructions.sh --repair-line`, so it alwa ## Regression coverage -`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the live-lock and fresh-beacon guard predicate, the cooperative `--claude` open-generation claim wait, monotonic failed-epoch progression, bounded attended fail-open, post-alarm continuation suppression, positive recovery reset, generation and legacy claim cases that must block or clear instead of allowing a blind stop, away-mode daemon ownership between watcher cycles and over a watcher lock left behind by an exited watcher, plus its dead, pid-reused, absent, stale-beacon, and away-mode-off negatives, Pi logical-run latching, missing-`jq` behavior, all five primary registrations, Grok native and legacy selection, typed field precedence, malformed input, and exactly-one-path safety. +`tests/fm-turnend-guard.test.sh` covers the predicate, main and secondmate primary scope, child-worktree exclusion, `FM_HOME` and `FM_STATE_OVERRIDE` precedence, the live-lock and fresh-beacon guard predicate, the cooperative `--claude` open-generation claim wait, monotonic failed-epoch progression, bounded attended fail-open, post-alarm continuation suppression, positive recovery reset, generation and legacy claim cases that must block or clear instead of allowing a blind stop, away-mode daemon ownership between watcher cycles and over a watcher lock left behind by an exited watcher, plus its dead, pid-reused, absent, stale-beacon, and away-mode-off negatives, the away-mode beacon's poll-derived grace widening for a live daemon still mid-cycle and its bound against a dead daemon, a beacon older than that wider grace, and FM_POLL's inapplicability with away mode off, Pi logical-run latching, missing-`jq` behavior, all five primary registrations, Grok native and legacy selection, typed field precedence, malformed input, and exactly-one-path safety. `tests/fm-guard-stale-banner.test.sh` covers the pull-guard predicate, including the persistent-model fresh-leftover-beacon negative control, the auto-arm model's healthy fresh-beacon-without-a-watcher case and stale-beacon alarm, and the extension model's live-watcher path, ownership-qualified fresh hand-off, held-lock failures, independently broken ownership signals, stale-beacon alarm, queued-wake warning, and Pi and pi-signed harness routing. It also covers true-reason banner wording and reason-keyed episode dedup surviving a beacon mtime change. `tests/fm-cursor-primary.test.sh` covers the Cursor park end to end over real processes with no harness installed: each tracked Claude-shaped entrypoint standing down on a Cursor payload, both follow-up sources, the bounded repair nag and its reset, the nested loop bounds, supersession, away-mode and lock-ownership inertness, Pi-host stand-down without Cursor identity and continued parking when `PI_CODING_AGENT` leaks alongside `CURSOR_AGENT` or `CURSOR_INVOKED_AS`, child-worktree exclusion, and that the adapter never exits 2. diff --git a/docs/verification/process-event-sources.md b/docs/verification/process-event-sources.md index 3ad75a65137..c37a0d89cf8 100644 --- a/docs/verification/process-event-sources.md +++ b/docs/verification/process-event-sources.md @@ -113,9 +113,13 @@ Exercised by `tests/fm-procevent.test.sh` against a fake blocking source whose c | one owner per canonical source | a second home's `start` for the same source id reports `already owned` and publishes nothing | | canonical physical identity | a final-component symlink and its target produce the same Lavish source id | | isolated public start boundary | direct `start` establishes a new runner-led process group before claiming the source, so retirement cannot signal an unrelated process inherited from the caller's group | +| guarded runner startup | the source command does not launch when the detached owner guard rejects an invalid lease configuration, proving the runner waits for positive guard readiness and fails closed when initialization fails | +| attached owner continuity | a foreground `start` with a one-second lease remains alive beyond that lease while its caller stays attached, then captures normally when the blocking source completes | +| owner-home lifetime and scope | a detached runner and its spawning descendant are observed reparented before an expired owner lease stops their whole process group and process churn; replacing the state directory at the same path cannot keep the old runner alive with a new lease because its recorded device/inode no longer matches, while an identical runner in an unchanged home whose reconcile cycle keeps its lease fresh remains alive | +| launch pacing during owner-loss grace | an immediately returning source that attempts detached self-relaunches is held to the configured minimum interval between command launches and remains bounded until its expired owner lease stops the generation; replacement starts a fresh pacing generation, prunes prior pacing state, and prevents a superseded sleeping runner from recreating it | | stale reclaim without displacement | concurrent contenders replacing one stale claim start exactly one runner, cross-home replacement removes the old generation's staging file from its recorded state directory, and a generation whose stale owner and independently empty process group prove it gone remains reclaimable when its recorded state-root identity can no longer be revalidated | -| crashed leader with a live owned group | `SIGKILL` on only the runner leader leaves its blocking child group alive; reconcile then stops that surviving group before any replacement starts, never leaves two source processes running for one canonical source, and a generation with no leader and no surviving group is still reclaimed | -| PID-reuse safety | retirement refuses to signal a live PID whose identity differs from the claim, a reused PID never reaches the group-stop path because its leader is alive, and a surviving old process group prevents stale-generation cleanup on both ordinary and failed reservation-removal paths | +| crashed leader with a live group | `SIGKILL` on only the runner leader leaves its blocking child group alive; reconcile treats that leaderless group as ambiguous, preserves its claim without starting a replacement, and still reclaims a generation with no leader and no surviving group | +| PID-reuse safety | retirement refuses a live PID whose identity differs from the claim before signalling, and a surviving process group prevents stale-generation cleanup on both ordinary and failed reservation-removal paths | | coherent ownership reads | a claim replacement held inside the source boundary blocks `list` until one complete generation is visible | | retire-start exclusion | a queued start revalidates registration after the serialized retirement boundary and executes no child | | uncertain identity | a live owner whose identity probe transiently fails is not signaled or released, and its registration remains for retry | @@ -173,20 +177,29 @@ The 2026-08-27 review inspected `bin/fm-harness.sh`, `bin/fm-supervision-instruc ## Runner lifetime and cleanup A runner started by `reconcile` is its own process group leader and is reparented to init, so it outlives the shell that started it by design. -That means nothing about the starting context can reap it: removing a home's state directory does not stop an already-running child, and signalling only the runner leaves the blocking child alive. +Removing a home's state directory does not stop an already-running child, and signalling only the runner leaves the blocking child alive. -Two paths therefore stop a runner, and both verify the runner-owned process group, escalate to `KILL` while that group still exists, and refuse to release ownership until the whole group is gone: +Three paths stop a runner generation through its verified process group: +- The runner starts only after its separate owner guard confirms initialization; the guard stops the runner group after two consecutive checks cannot prove the owning home's recorded physical identity and lease freshness. - `retire` resolves the runner PID and identity from this home's machine-wide claim, so retirement still works when the home's state is already gone. - `reconcile` stops a runner this home owns whose source registration has been removed, and reports it as `stopped=N`. +The owner guard and explicit cleanup paths reach the blocking source and its descendants through the runner's group. +The registration launch floor independently bounds repeated runner launches while an owner-loss lease is still valid. +The Lavish adapter's start-to-start poll governor separately bounds its internal retry loop under shipped defaults without delaying a normally blocking poll. +An attached public `start` maintains the lease for its caller's lifetime. +At the accepted confused-agent/accidental grade, the inherited `FM_PROCEVENT_IN_RUNNER` marker prevents detached runners and their ordinary children from refreshing it; adversarial unforgeability against a source that deliberately strips that marker is out of scope. + The same group rule decides when a claim may be reclaimed, not only when a runner may be signalled. -A leader that died while its owned group kept running is not a gone generation, so `reconcile` stops that surviving group and releases its generation before starting any replacement, and preserves the claim for a later retry when it cannot prove the group stopped or another home owns it. +A leader that died while its process group kept running is not a gone generation. +Because the leaderless group cannot be proved to belong to the recorded generation, `reconcile` preserves its claim without signalling it or starting a replacement. Once a stale owner and an independent group check prove the whole generation gone, an unreachable token-keyed capture reservation cannot veto reclamation. -Signalling an orphaned group is safe precisely because only an absent leader reaches that state: a reused PID leaves the leader alive, so no group signal follows, and an independent surviving-group check still prevents stale-generation cleanup. +Known limit: when either a live reused PID or an absent leader makes group ownership ambiguous, the reaper does not act because it cannot prove the group is the orphan generation; storm-rate containment plus ordinary lease and reconcile cleanup are the confused-agent-grade backstop. +Known limit: identity and process-group verification cannot be made atomic with signalling in portable shell. +The reaper signals only a target it has verified as the orphan generation, but PID and group reuse remain possible in the narrow interval between verification and the signal; launch pacing is the primary host-wedge protection and watchdog cleanup is a backstop. -This was found by four orphaned runners, elapsed 6-13 minutes, left by a suite whose fixture source never completed. -`tests/fm-procevent.test.sh` now covers both paths, and three consecutive suite runs leave zero runners, zero fixture children, and zero stray claims. +`tests/fm-procevent.test.sh` covers owner-loss reaping, descendant churn cessation, cross-home scope, launch pacing, guard startup failure, attached-start continuity, explicit retirement, and stale-group reconciliation. ## Portability finding diff --git a/docs/watcher-continuity.md b/docs/watcher-continuity.md index ebe3a7b413b..19d3d15f93d 100644 --- a/docs/watcher-continuity.md +++ b/docs/watcher-continuity.md @@ -72,6 +72,13 @@ Main records its presented set in `state/.main-eligible-rows`. A branch grant is published through `bin/fm-wake-grant.sh` under that same lock in `state/.branch-eligible-rows`, bound to the live branch process and extension generation recorded in `state/.branch-eligible-owner`, and publication is refused if main already claimed any requested row. A main drain validates that owner evidence under the queue lock and reclaims the grant when its process is gone or its identity no longer matches. A main drain claims every currently unclaimed row and excludes an active branch grant from both presentation and acknowledgement. +Because that exclusion makes those rows invisible to main, `bin/fm-guard.sh`'s queued-wake warning counts only the rows the calling actor can itself present or retire, so an actor is never sent to a drain that provably has nothing for it. +`bin/fm-wake-lib.sh` owns that per-actor count (`fm_wake_actor_pending_count`) alongside the grant row-list and owner-record reads that the drain and `bin/fm-wake-grant.sh` share. +A row a live grant reserves is therefore never counted as drainable for main; rather than going silent about a visibly non-empty queue, the guard prints a distinct advisory naming the live supervision branch as the holder and saying not to drain those rows from here. +The branch actor's queued-wake output stays suppressed in every case. +A main drain with nothing of its own left, and a live grant still holding the queue, says so in one bounded line instead of exiting silently. +A row that lost the five appended fields or its numeric sequence can never be claimed, presented, or named by an `--ack-through` cutoff, so a main drain retires it under the queue lock and reports how many it removed together with those rows verbatim, bounded to the first 20 and a count of the rest, because the queue was their only durable record; a branch drain never does, because a grant can only name sequences that were structurally valid when it was published. +A retirement that cannot be read or written is reported and never fails the drain: the rows that remain usable are still presented with their acknowledgement command, the unusable ones stay queued for a later drain to retire, and failing the whole drain would strand the usable rows too. Its `--ack-through ` deletes only claimed main rows at or below the cutoff, while a branch acknowledgement deletes only claimed branch rows at or below its cutoff. Every settled branch prompt releases any residual grant, so an omitted or failed acknowledgement leaves the durable row available to a later main drain; a successful acknowledgement has already removed it. An acknowledgement whose cutoff removes none of the actor's rows while a presented row above the cutoff still waits is reported as having acknowledged nothing, together with the exact `--ack-through` and `--recovery-generation` command for that presented row; the presented set is read before any re-claim, so a row that arrived after presentation is never named for unseen acknowledgement. @@ -82,6 +89,7 @@ A check-kind row is main-owned in every mode, including a heartbeat review, so i A missing or empty branch snapshot is refused loudly rather than read as "nothing eligible", because reaching the drain without the non-empty handoff promised by the extension is a wiring bug. Because branch claims contain no check-kind rows, a branch acknowledgement skips check-specific receipt scans. `tests/fm-wake-queue.test.sh`'s mixed-queue actor, stale-acknowledgement remedy, and presentation-deadline tests drive the real scripts: branch acknowledgement cannot swallow a main row, a concurrent main turn cannot present or acknowledge an active branch grant, a no-op stale acknowledgement names the current presented wake's exact command, live-holder presentation contention stays bounded and retriable, and acknowledgement locking remains blocking. +The same suite pins the counted-equals-presentable invariant against `bin/fm-guard.sh` and `bin/fm-wake-drain.sh` together: a branch-held row raises the held advisory rather than the ordinary queued-wake warning for main, and is presented with its acknowledgement command - with the ordinary warning restored - as soon as the grant clears, and structurally unusable rows are retired by main alone while every remaining row stays presentable and acknowledgeable. `tests/fm-pi-branch-extension.test.sh` pins extension-side classification, claim publication and release, and the pre-drain recheck. ## Arm-layer cycle contract diff --git a/tests/fm-backend-herdr-presentation-e2e.test.sh b/tests/fm-backend-herdr-presentation-e2e.test.sh index 8994bca1364..1dc6bd304f8 100755 --- a/tests/fm-backend-herdr-presentation-e2e.test.sh +++ b/tests/fm-backend-herdr-presentation-e2e.test.sh @@ -506,6 +506,7 @@ assert_no_projection_mutation_since() { # HOME_DIR="$TMP_ROOT/home" PROJECT_DIR="$TMP_ROOT/project" +RECOVERY_PROJECT_DIR="$TMP_ROOT/recovery-project" mkdir -p "$HOME_DIR/state" "$HOME_DIR/config" \ "$HOME_DIR/data/anchor" "$HOME_DIR/data/shape" \ "$HOME_DIR/data/order-a" "$HOME_DIR/data/order-b" \ @@ -530,6 +531,7 @@ write_ship_brief "$HOME_DIR" abort-b 'Projection abort fixture B.' write_ship_brief "$HOME_DIR" lock-contended 'Projection lock contention fixture.' write_ship_brief "$HOME_DIR" default-on 'Projection default-on fixture.' make_project "$PROJECT_DIR" +make_project "$RECOVERY_PROJECT_DIR" # Keep one ordinary primary task live so the durable firstmate workspace is # first and remains present while disposable workers are projected around it. @@ -1189,11 +1191,15 @@ teardown_task aflat "$SECOND_HOME_A" > "$TMP_ROOT/aflat-teardown.out" 2> "$TMP_R pass "real Herdr lab: session lock contention from a secondmate home falls back flat with no journal" # Same-identity recovery replaces only one exact agent-free husk in its -# original projected workspace. +# original projected workspace. These full-session restarts also stop the +# earlier multi-home workers whose restored panes are retained for the final +# exact-pane cleanup assertions. Keep the recovery fixtures in their own +# Treehouse pool so those intentionally retained records cannot claim a slot +# that a recovery fixture legitimately acquires after their processes stop. # Exercise both the leading fm- identity style seen in Hi Bit work and the # project-name identity style used by Wheelhouse work. for RESTART_ID in fm-hibit-resume-r1 wheelhouse-healing-r1; do - spawn_task "$RESTART_ID" "$HOME_DIR" "$PROJECT_DIR" > "$TMP_ROOT/$RESTART_ID-first.out" 2> "$TMP_ROOT/$RESTART_ID-first.err" \ + spawn_task "$RESTART_ID" "$HOME_DIR" "$RECOVERY_PROJECT_DIR" > "$TMP_ROOT/$RESTART_ID-first.out" 2> "$TMP_ROOT/$RESTART_ID-first.err" \ || fail "$RESTART_ID fixture's projected spawn failed: $(cat "$TMP_ROOT/$RESTART_ID-first.err")" RESTART_META="$HOME_DIR/state/$RESTART_ID.meta" OLD_RESTART_WT=$(remember_meta_worktree "$RESTART_META") @@ -1223,7 +1229,7 @@ for RESTART_ID in fm-hibit-resume-r1 wheelhouse-healing-r1; do fail "$RESTART_ID restart fixture unexpectedly retained a registered agent" fi RECLAIM_FOCUS=$(focus_snapshot) - spawn_task "$RESTART_ID" "$HOME_DIR" "$PROJECT_DIR" > "$TMP_ROOT/$RESTART_ID-reclaim.out" 2> "$TMP_ROOT/$RESTART_ID-reclaim.err" \ + spawn_task "$RESTART_ID" "$HOME_DIR" "$RECOVERY_PROJECT_DIR" > "$TMP_ROOT/$RESTART_ID-reclaim.out" 2> "$TMP_ROOT/$RESTART_ID-reclaim.err" \ || fail "$RESTART_ID same-identity reclaim failed: $(cat "$TMP_ROOT/$RESTART_ID-reclaim.err")" NEW_RESTART_WT=$(remember_meta_worktree "$RESTART_META") NEW_RESTART_WSID=$(grep '^herdr_workspace_id=' "$RESTART_META" | cut -d= -f2-) @@ -1248,7 +1254,7 @@ for RESTART_ID in fm-hibit-resume-r1 wheelhouse-healing-r1; do || fail "could not reprovision the isolated session for idempotent reclaim" PRIOR_RESTART_WT=$NEW_RESTART_WT PRIOR_RESTART_PANE=$NEW_RESTART_PANE - spawn_task "$RESTART_ID" "$HOME_DIR" "$PROJECT_DIR" > "$TMP_ROOT/$RESTART_ID-idempotent.out" 2> "$TMP_ROOT/$RESTART_ID-idempotent.err" \ + spawn_task "$RESTART_ID" "$HOME_DIR" "$RECOVERY_PROJECT_DIR" > "$TMP_ROOT/$RESTART_ID-idempotent.out" 2> "$TMP_ROOT/$RESTART_ID-idempotent.err" \ || fail "$RESTART_ID repeated reclaim failed: $(cat "$TMP_ROOT/$RESTART_ID-idempotent.err")" NEW_RESTART_WT=$(remember_meta_worktree "$RESTART_META") NEW_RESTART_WSID=$(grep '^herdr_workspace_id=' "$RESTART_META" | cut -d= -f2-) @@ -1275,7 +1281,7 @@ pass "real Herdr lab: Hi Bit and Wheelhouse-style same-identity restarts reclaim CROSS_RESTART_ID=wheel-child-resume mkdir -p "$SECOND_HOME_A/data/$CROSS_RESTART_ID" write_ship_brief "$SECOND_HOME_A" "$CROSS_RESTART_ID" 'Cross-home restart fixture.' -spawn_task "$CROSS_RESTART_ID" "$SECOND_HOME_A" "$PROJECT_DIR" > "$TMP_ROOT/cross-restart-first.out" 2> "$TMP_ROOT/cross-restart-first.err" \ +spawn_task "$CROSS_RESTART_ID" "$SECOND_HOME_A" "$RECOVERY_PROJECT_DIR" > "$TMP_ROOT/cross-restart-first.out" 2> "$TMP_ROOT/cross-restart-first.err" \ || fail "cross-home restart fixture failed: $(cat "$TMP_ROOT/cross-restart-first.err")" CROSS_RESTART_META="$SECOND_HOME_A/state/$CROSS_RESTART_ID.meta" CROSS_OLD_WT=$(remember_meta_worktree "$CROSS_RESTART_META") @@ -1291,7 +1297,7 @@ PATH="$HERDR_ORIGINAL_PATH" "$HERDR_LAB_HELPER" stop "$HERDR_LAB_SESSION" >/dev/ || fail "could not stop the isolated session for cross-home restart" PATH="$HERDR_ORIGINAL_PATH" "$HERDR_LAB_HELPER" provision "$HERDR_LAB_SESSION" \ || fail "could not reprovision the isolated session for cross-home restart" -spawn_task "$CROSS_RESTART_ID" "$SECOND_HOME_A" "$PROJECT_DIR" > "$TMP_ROOT/cross-restart-resume.out" 2> "$TMP_ROOT/cross-restart-resume.err" \ +spawn_task "$CROSS_RESTART_ID" "$SECOND_HOME_A" "$RECOVERY_PROJECT_DIR" > "$TMP_ROOT/cross-restart-resume.out" 2> "$TMP_ROOT/cross-restart-resume.err" \ || fail "cross-home same-identity reclaim failed: $(cat "$TMP_ROOT/cross-restart-resume.err")" CROSS_NEW_WT=$(remember_meta_worktree "$CROSS_RESTART_META") CROSS_NEW_WSID=$(grep '^herdr_workspace_id=' "$CROSS_RESTART_META" | cut -d= -f2-) @@ -1313,9 +1319,9 @@ BRAVO_WAVE_ID=resume-wave-bravo mkdir -p "$HOME_DIR/data/$PRIMARY_WAVE_ID" "$SECOND_HOME_B/data/$BRAVO_WAVE_ID" write_ship_brief "$HOME_DIR" "$PRIMARY_WAVE_ID" 'Concurrent primary recovery fixture.' write_ship_brief "$SECOND_HOME_B" "$BRAVO_WAVE_ID" 'Concurrent secondmate recovery fixture.' -spawn_task "$PRIMARY_WAVE_ID" "$HOME_DIR" "$PROJECT_DIR" > "$TMP_ROOT/primary-wave-first.out" 2> "$TMP_ROOT/primary-wave-first.err" \ +spawn_task "$PRIMARY_WAVE_ID" "$HOME_DIR" "$RECOVERY_PROJECT_DIR" > "$TMP_ROOT/primary-wave-first.out" 2> "$TMP_ROOT/primary-wave-first.err" \ || fail "primary recovery-wave fixture failed: $(cat "$TMP_ROOT/primary-wave-first.err")" -spawn_task "$BRAVO_WAVE_ID" "$SECOND_HOME_B" "$PROJECT_DIR" > "$TMP_ROOT/bravo-wave-first.out" 2> "$TMP_ROOT/bravo-wave-first.err" \ +spawn_task "$BRAVO_WAVE_ID" "$SECOND_HOME_B" "$RECOVERY_PROJECT_DIR" > "$TMP_ROOT/bravo-wave-first.out" 2> "$TMP_ROOT/bravo-wave-first.err" \ || fail "secondmate recovery-wave fixture failed: $(cat "$TMP_ROOT/bravo-wave-first.err")" PRIMARY_WAVE_META="$HOME_DIR/state/$PRIMARY_WAVE_ID.meta" BRAVO_WAVE_META="$SECOND_HOME_B/state/$BRAVO_WAVE_ID.meta" @@ -1330,9 +1336,9 @@ PATH="$HERDR_ORIGINAL_PATH" "$HERDR_LAB_HELPER" stop "$HERDR_LAB_SESSION" >/dev/ PATH="$HERDR_ORIGINAL_PATH" "$HERDR_LAB_HELPER" provision "$HERDR_LAB_SESSION" \ || fail "could not reprovision the isolated session for concurrent recovery" CONCURRENT_RECOVERY_FOCUS=$(focus_snapshot) -spawn_task "$PRIMARY_WAVE_ID" "$HOME_DIR" "$PROJECT_DIR" > "$TMP_ROOT/primary-wave-resume.out" 2> "$TMP_ROOT/primary-wave-resume.err" & +spawn_task "$PRIMARY_WAVE_ID" "$HOME_DIR" "$RECOVERY_PROJECT_DIR" > "$TMP_ROOT/primary-wave-resume.out" 2> "$TMP_ROOT/primary-wave-resume.err" & PRIMARY_WAVE_PID=$! -spawn_task "$BRAVO_WAVE_ID" "$SECOND_HOME_B" "$PROJECT_DIR" > "$TMP_ROOT/bravo-wave-resume.out" 2> "$TMP_ROOT/bravo-wave-resume.err" & +spawn_task "$BRAVO_WAVE_ID" "$SECOND_HOME_B" "$RECOVERY_PROJECT_DIR" > "$TMP_ROOT/bravo-wave-resume.out" 2> "$TMP_ROOT/bravo-wave-resume.err" & BRAVO_WAVE_PID=$! wait "$PRIMARY_WAVE_PID" || fail "concurrent primary recovery failed: $(cat "$TMP_ROOT/primary-wave-resume.err")" wait "$BRAVO_WAVE_PID" || fail "concurrent secondmate recovery failed: $(cat "$TMP_ROOT/bravo-wave-resume.err")" diff --git a/tests/fm-backend-orca.test.sh b/tests/fm-backend-orca.test.sh index aa8b58aeb6a..60564719239 100755 --- a/tests/fm-backend-orca.test.sh +++ b/tests/fm-backend-orca.test.sh @@ -539,7 +539,7 @@ test_spawn_writes_orca_metadata_and_launches_harness() { "spawn should reuse the implicit terminal returned by Orca worktree creation" assert_contains "$(cat "$log")" $'orca\x1f''terminal'$'\x1f''send'$'\x1f''--terminal'$'\x1f''term-spawn'$'\x1f''--text'$'\x1f''export GOTMPDIR=/tmp/fm-orcaspawnz1/gotmp'$'\x1f''--enter'$'\x1f''--json' \ "spawn did not export GOTMPDIR through the Orca terminal" - assert_contains "$(cat "$log")" "CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false CLAUDE_CODE_SEND_FEEDBACK=0 claude --dangerously-skip-permissions --settings '{\"feedbackDrafts\":\"off\"}'" \ + assert_contains "$(cat "$log")" "CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false CLAUDE_CODE_SEND_FEEDBACK=0 claude --dangerously-skip-permissions --settings '{\"feedbackDrafts\":\"off\",\"attribution\":{\"commit\":\"\",\"pr\":\"\",\"sessionUrl\":false}}'" \ "spawn did not send the selected harness launch command through Orca" rm -rf "/tmp/fm-$id" pass "fm-spawn.sh --backend orca: reuses implicit terminal, records metadata, launches harness" diff --git a/tests/fm-backlog-atomicity.test.sh b/tests/fm-backlog-atomicity.test.sh index ec45f93399f..d8f6d1bfe9c 100755 --- a/tests/fm-backlog-atomicity.test.sh +++ b/tests/fm-backlog-atomicity.test.sh @@ -2366,6 +2366,13 @@ test_bootstrap_refuses_a_symlinked_state_directory_before_reconciliation() { test_bootstrap_stops_when_data_disappears_before_reconciliation() { local case_dir id saved out rc=0 + # The data-removal fault is injected by a fake stat on PATH; on Darwin the + # budget link-count helper now calls /usr/bin/stat directly, so the fake can + # never fire there. Skip the Darwin run of this case. + if [ "$(uname)" = Darwin ]; then + pass "bootstrap data-disappears fault injection is PATH-based; skipped on Darwin where stat is /usr/bin/stat" + return + fi id=atomic-bootstrap-data-race-b11 case_dir=$(make_home bootstrap-data-race) add_item "$case_dir" "$id" diff --git a/tests/fm-bearings-board-render.test.sh b/tests/fm-bearings-board-render.test.sh index f1ac5fe8e6f..cf26fd31428 100755 --- a/tests/fm-bearings-board-render.test.sh +++ b/tests/fm-bearings-board-render.test.sh @@ -18,24 +18,12 @@ TMP_ROOT=$(fm_test_tmproot fm-bearings-board-render) command -v jq >/dev/null 2>&1 || { echo "skip: jq not found"; exit 0; } command -v node >/dev/null 2>&1 || { echo "skip: node not found"; exit 0; } -# A build starts a listener for the board it publishes, so every home this -# suite creates is swept before the fixture directory is removed. -RENDER_HOMES=() - -render_teardown() { - local home - for home in ${RENDER_HOMES[@]+"${RENDER_HOMES[@]}"}; do - FM_HOME="$home" FM_STATE_OVERRIDE="$home/state" FM_DATA_OVERRIDE="$home/data" \ - FM_PROCEVENT_CLAIM_ROOT="$home/procevent-claims" \ - "$ROOT/bin/fm-procevent.sh" sweep-home >/dev/null 2>&1 || true - done - fm_test_cleanup -} -trap render_teardown EXIT - make_home() { # local home="$TMP_ROOT/$1" fakebin - RENDER_HOMES+=("$home") + # A build starts a listener for the board it publishes. Registered with + # tests/lib.sh, not with a shell array: make_home runs inside a command + # substitution, where an array append never reaches the caller. + fm_test_track_procevent_home "$home" "$home/procevent-claims" mkdir -p "$home/state" "$home/data" fakebin=$(fm_fakebin "$home") # The build proves the board session is live before it arms anything, so the @@ -51,7 +39,11 @@ case "${1-}" in [ ! -s "$FM_HOME/lavish-open" ] \ || printf ' %s,open,"http://127.0.0.1/session/render",0\n' "$(cat "$FM_HOME/lavish-open")" ;; - poll) while :; do sleep 1; done ;; + poll) + # Bounded, so a listener that escapes its test stops on its own. + while [ "$SECONDS" -lt "${FM_TEST_STUB_MAX_BLOCK_SECONDS:-120}" ]; do sleep 1; done + exit 75 + ;; *) real=$(cd "$(dirname "$1")" && pwd -P)/$(basename "$1") printf '%s\n' "$real" > "$FM_HOME/lavish-open" diff --git a/tests/fm-bearings-board.test.sh b/tests/fm-bearings-board.test.sh index 7838518459e..b59010036e9 100644 --- a/tests/fm-bearings-board.test.sh +++ b/tests/fm-bearings-board.test.sh @@ -13,22 +13,6 @@ TMP_ROOT=$(fm_test_tmproot fm-bearings-board) command -v jq >/dev/null 2>&1 || { echo "skip: jq not found"; exit 0; } -# Every home this suite creates, so teardown can stop the listeners its builds -# start. A detached runner is reparented, so removing the fixture directory -# does not stop an already-running child. -BOARD_HOMES=() - -board_teardown() { - local home - for home in ${BOARD_HOMES[@]+"${BOARD_HOMES[@]}"}; do - FM_HOME="$home" FM_STATE_OVERRIDE="$home/state" FM_DATA_OVERRIDE="$home/data" \ - FM_PROCEVENT_CLAIM_ROOT="$home/procevent-claims" \ - "$ROOT/bin/fm-procevent.sh" sweep-home >/dev/null 2>&1 || true - done - fm_test_cleanup -} -trap board_teardown EXIT - # A lavish-axi stub that reproduces the shapes verified against the real # lavish-axi 0.1.61, because the build's liveness verdict is read from what the # vendor emits. The load-bearing shape is the refusal: opening a session the @@ -38,7 +22,10 @@ trap board_teardown EXIT # plain open refuse, and `refuse-reopen` makes even --reopen leave it dead. make_home() { # local home="$TMP_ROOT/$1" fakebin - BOARD_HOMES+=("$home") + # Registered with tests/lib.sh, not with a shell array: make_home is called + # inside a command substitution, so an array append here never reaches the + # caller and every listener this suite started used to survive the run. + fm_test_track_procevent_home "$home" "$home/procevent-claims" mkdir -p "$home/state" "$home/data" "$home/lavish-state" fakebin=$(fm_fakebin "$home") cat > "$fakebin/lavish-axi" <<'SH' @@ -56,11 +43,20 @@ case "${1-}" in poll) # A real blocking listener: it returns only when the trigger appears, so a # live owner in these tests is a live process rather than a timing artifact. - while [ ! -e "$state/poll-trigger" ]; do sleep 0.05; done + # Both waits are bounded, so a listener that escapes its test cannot keep + # spawning processes for as long as the host stays up. + limit=${FM_TEST_STUB_MAX_BLOCK_SECONDS:-120} + while [ ! -e "$state/poll-trigger" ]; do + [ "$SECONDS" -lt "$limit" ] || exit 75 + sleep 0.05 + done printf 'session:\n status: ended\n' if [ -e "$state/hold-after-terminal" ]; then : > "$state/terminal-emitted" - while [ -e "$state/hold-after-terminal" ]; do sleep 0.05; done + while [ -e "$state/hold-after-terminal" ]; do + [ "$SECONDS" -lt "$limit" ] || exit 75 + sleep 0.05 + done fi exit 0 ;; diff --git a/tests/fm-bootstrap.test.sh b/tests/fm-bootstrap.test.sh index 3423cc4acfc..b25d9c69899 100755 --- a/tests/fm-bootstrap.test.sh +++ b/tests/fm-bootstrap.test.sh @@ -519,7 +519,7 @@ test_orca_backend_gates_orca_tool_only_when_selected() { printf '%s\n' manual > "$case_dir/home/config/backlog-backend" printf '%s\n' orca > "$case_dir/home/config/backend" fakebin=$(make_fake_toolchain "$case_dir") - out=$(PATH="$fakebin:$BASE_PATH" FM_HOME="$case_dir/home" FM_ROOT_OVERRIDE="$case_dir/home" \ + out=$(PATH="$fakebin:$(fm_test_base_path_sans "$BASE_PATH" orca)" FM_HOME="$case_dir/home" FM_ROOT_OVERRIDE="$case_dir/home" \ FM_FAKE_TREEHOUSE_LEASE_HELP=1 "$ROOT/bin/fm-bootstrap.sh") [ "$out" = "$missing_orca" ] || fail "backend=orca should require only the Orca-specific missing tool, got: $out" @@ -901,17 +901,17 @@ exit 1 SH chmod +x "$fakebin/gh" - all_out=$(PATH="$fakebin:$BASE_PATH" FM_HOME="$case_dir/home" FM_ROOT_OVERRIDE="$case_dir/home" \ + all_out=$(PATH="$fakebin:$(fm_test_base_path_sans "$BASE_PATH" node)" FM_HOME="$case_dir/home" FM_ROOT_OVERRIDE="$case_dir/home" \ FM_FAKE_TREEHOUSE_LEASE_HELP=1 "$ROOT/bin/fm-bootstrap.sh") assert_contains "$all_out" "MISSING: node (install:" "the unsplit run lost its local diagnostic" assert_contains "$all_out" "NEEDS_GH_AUTH" "the unsplit run lost its network diagnostic" - skip_out=$(PATH="$fakebin:$BASE_PATH" FM_HOME="$case_dir/home" FM_ROOT_OVERRIDE="$case_dir/home" \ + skip_out=$(PATH="$fakebin:$(fm_test_base_path_sans "$BASE_PATH" node)" FM_HOME="$case_dir/home" FM_ROOT_OVERRIDE="$case_dir/home" \ FM_FAKE_TREEHOUSE_LEASE_HELP=1 FM_BOOTSTRAP_NETWORK=skip "$ROOT/bin/fm-bootstrap.sh") assert_contains "$skip_out" "MISSING: node (install:" "the local half lost its own diagnostic" assert_not_contains "$skip_out" "NEEDS_GH_AUTH" "the local half still made a network call" - only_out=$(PATH="$fakebin:$BASE_PATH" FM_HOME="$case_dir/home" FM_ROOT_OVERRIDE="$case_dir/home" \ + only_out=$(PATH="$fakebin:$(fm_test_base_path_sans "$BASE_PATH" node)" FM_HOME="$case_dir/home" FM_ROOT_OVERRIDE="$case_dir/home" \ FM_FAKE_TREEHOUSE_LEASE_HELP=1 FM_BOOTSTRAP_NETWORK=only "$ROOT/bin/fm-bootstrap.sh") assert_contains "$only_out" "NEEDS_GH_AUTH" "the network half lost its own diagnostic" assert_not_contains "$only_out" "MISSING: node" "the network half repeated the local half's work" @@ -922,7 +922,7 @@ SH # A typo must never silently drop a safety sweep, so anything unrecognized # resolves to the complete run. - [ "$(PATH="$fakebin:$BASE_PATH" FM_HOME="$case_dir/home" FM_ROOT_OVERRIDE="$case_dir/home" \ + [ "$(PATH="$fakebin:$(fm_test_base_path_sans "$BASE_PATH" node)" FM_HOME="$case_dir/home" FM_ROOT_OVERRIDE="$case_dir/home" \ FM_FAKE_TREEHOUSE_LEASE_HELP=1 FM_BOOTSTRAP_NETWORK=sikp "$ROOT/bin/fm-bootstrap.sh")" = "$all_out" ] \ || fail "an unrecognized FM_BOOTSTRAP_NETWORK value did not fall back to the complete run" pass "bootstrap: FM_BOOTSTRAP_NETWORK partitions one run into local and network halves" diff --git a/tests/fm-calm-pi-extension.test.sh b/tests/fm-calm-pi-extension.test.sh index 2c6c404b229..b873a9b1b61 100755 --- a/tests/fm-calm-pi-extension.test.sh +++ b/tests/fm-calm-pi-extension.test.sh @@ -71,6 +71,82 @@ find_chrome() { return 1 } +# Render an exported session in real Chrome and leave the DOM in . +# +# Rendering is a vendor-tool step, not a Calm guarantee: the DOM assertions the +# caller runs afterwards are what protect the contract. Headless Chrome start-up +# is the part that fails intermittently on a loaded CI runner - it can exit +# before writing any DOM at all - and the original single unattended attempt +# discarded both Chrome's stderr and its exit status, so a CI break surfaced as +# a bare "could not render" with nothing in the log to tell a Chrome start-up +# crash apart from a real change in Pi's export shape. +# +# So: retry the render a bounded number of times on a fresh profile, and when +# every attempt fails, print the Chrome binary, its version, the installed Pi +# version, and each attempt's exit status, stderr tail, and whether the helper +# timed the attempt out - when it did, the exit status is only this helper's own +# kill signal. The extra flags remove Chrome's background-network and /dev/shm +# dependencies, which are the start-up surfaces that fail on a runner; neither +# changes the rendered DOM of a local file. +render_export_dom() { + local chrome=$1 source_file=$2 out_file=$3 pi_version=$4 + local attempt pid status wait_count wait_limit reap_wait log profile report timed_out + report="$TMP_ROOT/chrome-render-report.txt" + wait_limit=${FM_CHROME_RENDER_WAIT_TICKS:-300} + : >"$report" + for attempt in 1 2 3; do + log="$TMP_ROOT/chrome-render-$attempt.err" + profile="$TMP_ROOT/chrome-profile-$attempt" + rm -rf "$profile" + : >"$out_file" + "$chrome" \ + --headless=new \ + --disable-gpu \ + --no-sandbox \ + --disable-dev-shm-usage \ + --disable-background-networking \ + --user-data-dir="$profile" \ + --virtual-time-budget=2000 \ + --dump-dom \ + "file://$source_file" >"$out_file" 2>"$log" & + pid=$! + # Check the DOM before Chrome's liveness, so an attempt that writes the + # complete dump and exits immediately is still read as a success. + wait_count=0 + while [ "$wait_count" -lt "$wait_limit" ]; do + grep -Fq '' "$out_file" 2>/dev/null && break + kill -0 "$pid" 2>/dev/null || break + sleep 0.1 + wait_count=$((wait_count + 1)) + done + timed_out=no + if [ "$wait_count" -ge "$wait_limit" ]; then + timed_out=yes + fi + kill "$pid" 2>/dev/null || true + # Chrome can retain --headless=new after --dump-dom completes and ignore TERM, + # so an unbounded wait can hang after the complete DOM has been captured. + reap_wait=0 + while kill -0 "$pid" 2>/dev/null && [ "$reap_wait" -lt 20 ]; do + sleep 0.1 + reap_wait=$((reap_wait + 1)) + done + if kill -0 "$pid" 2>/dev/null; then + kill -9 "$pid" 2>/dev/null || true + fi + status=0 + wait "$pid" 2>/dev/null || status=$? + grep -Fq '' "$out_file" 2>/dev/null && return 0 + printf 'attempt %s: exit=%s timed_out=%s bytes=%s stderr=%s\n' \ + "$attempt" "$status" "$timed_out" "$(wc -c <"$out_file" | tr -d ' ')" \ + "$(tail -c 400 "$log" 2>/dev/null | tr '\n' ' ')" >>"$report" + done + printf 'chrome=%s chrome_version=%s pi=%s; %s' \ + "$chrome" "$("$chrome" --version 2>&1 | head -1)" "$pi_version" \ + "$(tr '\n' ' ' <"$report")" + return 1 +} + test_home_resolution() { local fixture out status version if ! command -v node >/dev/null 2>&1 || ! command -v npm >/dev/null 2>&1; then @@ -3107,8 +3183,105 @@ JS pass "Pi Calm working ship moves on a slow independent cadence over faster fixed-cell blue water, paints the complete boat standard yellow with balanced resets, keeps ANSI-stripped width exact, flips the directional sail on the exact bounce at both edges and every width, clamps visible and hidden resizes, falls back deterministically when narrow, freezes and resumes column/direction across settle/start without hidden-time jumps or duplicate timers, resets only on a fresh session, and installs and removes one scheduler-owning widget across starts, settle, abort, failure, shutdown, reload, replacement, and Calm toggles while leaving Calm-off visibility untouched" } +# The rendered-DOM assertions below depend on a real browser, so the render step +# itself is the part that fails for reasons that have nothing to do with Calm. +# This pins that guard with real processes and no browser: one clean render, one +# that only succeeds after Chrome's start-up flake, and one that never renders +# and must report enough to tell a Chrome failure apart from a Pi export change. +test_export_dom_render_guard() { + local dir source_file out_file report + + dir="$TMP_ROOT/render-guard" + mkdir -p "$dir" + source_file="$dir/export.html" + out_file="$dir/dom.html" + printf 'export\n' >"$source_file" + + cat >"$dir/chrome-ok" <<'SH' +#!/bin/sh +case "${1:-}" in --version) echo "FakeChrome 1.2.3"; exit 0 ;; esac +echo attempt >>"$FM_FAKE_CHROME_ATTEMPTS" +printf 'export\n' +SH + cat >"$dir/chrome-flaky" <<'SH' +#!/bin/sh +case "${1:-}" in --version) echo "FakeChrome 1.2.3"; exit 0 ;; esac +echo attempt >>"$FM_FAKE_CHROME_ATTEMPTS" +if [ "$(wc -l <"$FM_FAKE_CHROME_ATTEMPTS")" -lt 3 ]; then + echo "fake chrome start-up crashed" >&2 + exit 1 +fi +printf 'export\n' +SH + cat >"$dir/chrome-broken" <<'SH' +#!/bin/sh +case "${1:-}" in --version) echo "FakeChrome 1.2.3"; exit 0 ;; esac +echo attempt >>"$FM_FAKE_CHROME_ATTEMPTS" +echo "FAKE_CHROME_STARTUP_MARKER" >&2 +exit 9 +SH + cat >"$dir/chrome-hang" <<'SH' +#!/bin/sh +case "${1:-}" in --version) echo "FakeChrome 1.2.3"; exit 0 ;; esac +echo attempt >>"$FM_FAKE_CHROME_ATTEMPTS" +printf 'export' +exec sleep 30 +SH + chmod +x "$dir/chrome-ok" "$dir/chrome-flaky" "$dir/chrome-broken" "$dir/chrome-hang" + + : >"$dir/attempts-ok" + FM_FAKE_CHROME_ATTEMPTS="$dir/attempts-ok" \ + render_export_dom "$dir/chrome-ok" "$source_file" "$out_file" 9.9.9 >"$dir/report-ok" \ + || fail "render_export_dom rejected a Chrome that dumped a complete DOM" + grep -Fq '' "$out_file" || fail "render_export_dom did not leave the rendered DOM behind" + [ "$(wc -l <"$dir/attempts-ok")" -eq 1 ] \ + || fail "render_export_dom retried a Chrome that had already rendered the DOM" + [ ! -s "$dir/report-ok" ] || fail "render_export_dom reported a diagnostic for a successful render" + + : >"$dir/attempts-flaky" + : >"$out_file" + FM_FAKE_CHROME_ATTEMPTS="$dir/attempts-flaky" \ + render_export_dom "$dir/chrome-flaky" "$source_file" "$out_file" 9.9.9 >"$dir/report-flaky" \ + || fail "render_export_dom gave up on a Chrome that renders after a start-up failure" + grep -Fq '' "$out_file" || fail "a retried render left no DOM behind" + [ "$(wc -l <"$dir/attempts-flaky")" -eq 3 ] \ + || fail "render_export_dom did not retry the failed Chrome start-ups exactly" + + : >"$dir/attempts-broken" + : >"$out_file" + if FM_FAKE_CHROME_ATTEMPTS="$dir/attempts-broken" \ + render_export_dom "$dir/chrome-broken" "$source_file" "$out_file" 9.9.9 >"$dir/report-broken" + then + fail "render_export_dom accepted a Chrome that never rendered the DOM" + fi + [ "$(wc -l <"$dir/attempts-broken")" -eq 3 ] \ + || fail "render_export_dom did not exhaust its bounded retries before failing" + report=$(cat "$dir/report-broken") + assert_contains "$report" "$dir/chrome-broken" "the render failure did not name the Chrome binary it used" + assert_contains "$report" "FakeChrome 1.2.3" "the render failure did not name the Chrome version it used" + assert_contains "$report" "pi=9.9.9" "the render failure did not name the installed Pi version" + assert_contains "$report" "exit=9" "the render failure did not report Chrome's exit status" + assert_contains "$report" "timed_out=no" "the render failure did not report that Chrome exited on its own" + assert_contains "$report" "FAKE_CHROME_STARTUP_MARKER" "the render failure discarded Chrome's own diagnostic" + + : >"$dir/attempts-hang" + : >"$out_file" + if FM_FAKE_CHROME_ATTEMPTS="$dir/attempts-hang" FM_CHROME_RENDER_WAIT_TICKS=3 \ + render_export_dom "$dir/chrome-hang" "$source_file" "$out_file" 9.9.9 >"$dir/report-hang" + then + fail "render_export_dom accepted a Chrome that never finished the DOM" + fi + [ "$(wc -l <"$dir/attempts-hang")" -eq 3 ] \ + || fail "render_export_dom did not exhaust its bounded retries on a Chrome that never finished" + report=$(cat "$dir/report-hang") + assert_contains "$report" "timed_out=yes" \ + "the render failure reported its own kill signal without saying the attempt was timed out" + + pass "the rendered-export-DOM guard renders in one pass, retries a bounded number of Chrome start-up failures, and reports the Chrome binary, Chrome version, Pi version, exit status, and Chrome diagnostic when every attempt fails" +} + test_interactive_terminal_e2e() { - local project config home session_file export_file export_dom default_snapshot expanded_snapshot hidden_snapshot active_before_snapshot active_hidden_snapshot export_snapshot export_settled_snapshot restored_snapshot working_snapshot working_response_snapshot restarted_snapshot resumed_restored_snapshot hash_before hash_after now version chrome chrome_pid chrome_wait chrome_reap_wait active_wait active_screen_wait boat_frame_one boat_frame_two boat_resized_snapshot boat_focus_snapshot boat_cleared_snapshot boat_hull_line boat_sail_line boat_column_one boat_column_two boat_line boat_color_snapshot boat_color_line boat_water_snapshot boat_water_line boat_water_first boat_water_changed boat_narrow_snapshot boat_narrow_sails boat_freeze_snapshot boat_resume_snapshot boat_freeze_column boat_freeze_sail boat_resume_column boat_resume_sail + local project config home session_file export_file export_dom default_snapshot expanded_snapshot hidden_snapshot active_before_snapshot active_hidden_snapshot export_snapshot export_settled_snapshot restored_snapshot working_snapshot working_response_snapshot restarted_snapshot resumed_restored_snapshot hash_before hash_after now version chrome chrome_report active_wait active_screen_wait boat_frame_one boat_frame_two boat_resized_snapshot boat_focus_snapshot boat_cleared_snapshot boat_hull_line boat_sail_line boat_column_one boat_column_two boat_line boat_color_snapshot boat_color_line boat_water_snapshot boat_water_line boat_water_first boat_water_changed boat_narrow_snapshot boat_narrow_sails boat_freeze_snapshot boat_resume_snapshot boat_freeze_column boat_freeze_sail boat_resume_column boat_resume_sail if ! command -v pi >/dev/null 2>&1 || ! command -v tmux >/dev/null 2>&1; then echo "skip: pi or tmux not found for Pi calm interactive E2E" return 0 @@ -3554,36 +3727,10 @@ if (!serialized.includes("firstmate-synthetic-input") || !serialized.includes("/ const synthetic = entries.find((entry) => entry.type === "custom_message" && entry.customType === "firstmate-synthetic-input"); if (!synthetic || synthetic.display) process.exit(1); JS - chrome=$(find_chrome) || fail "Chrome or Chromium is required for rendered export DOM assertions" - "$chrome" \ - --headless=new \ - --disable-gpu \ - --no-sandbox \ - --user-data-dir="$TMP_ROOT/chrome-profile" \ - --virtual-time-budget=2000 \ - --dump-dom \ - "file://$export_file" >"$export_dom" 2>/dev/null & - chrome_pid=$! - chrome_wait=0 - while kill -0 "$chrome_pid" 2>/dev/null && [ "$chrome_wait" -lt 100 ]; do - grep -Fq '' "$export_dom" 2>/dev/null && break - sleep 0.1 - chrome_wait=$((chrome_wait + 1)) - done - kill "$chrome_pid" 2>/dev/null || true - # Chrome can retain --headless=new after --dump-dom completes and ignore TERM, - # so an unbounded wait can hang after the complete DOM has been captured. - chrome_reap_wait=0 - while kill -0 "$chrome_pid" 2>/dev/null && [ "$chrome_reap_wait" -lt 20 ]; do - sleep 0.1 - chrome_reap_wait=$((chrome_reap_wait + 1)) - done - if kill -0 "$chrome_pid" 2>/dev/null; then - kill -9 "$chrome_pid" 2>/dev/null || true - fi - wait "$chrome_pid" 2>/dev/null || true - grep -Fq '' "$export_dom" 2>/dev/null \ - || fail "could not render calm-mode HTML export DOM" + chrome=$(find_chrome) \ + || fail "Chrome or Chromium is required for rendered export DOM assertions; set FM_CHROME_BIN to one" + chrome_report=$(render_export_dom "$chrome" "$export_file" "$export_dom" "$version") \ + || fail "could not render calm-mode HTML export DOM: $chrome_report" node - "$export_dom" <<'JS' || fail "rendered export DOM violated the Calm conversation boundary" const dom = require("node:fs").readFileSync(process.argv[2], "utf8"); const messages = dom.match(/
([\s\S]*?)<\/main>/)?.[1]; @@ -4017,4 +4164,5 @@ test_calm_mid_turn_working_notes test_operational_followup_turn_e2e test_hidden_block_geometry_e2e test_working_ship_geometry_and_lifecycle +test_export_dom_render_guard test_interactive_terminal_e2e diff --git a/tests/fm-claude-stop-autoarm.test.sh b/tests/fm-claude-stop-autoarm.test.sh index 6917d21aac7..bbae5f16574 100755 --- a/tests/fm-claude-stop-autoarm.test.sh +++ b/tests/fm-claude-stop-autoarm.test.sh @@ -174,6 +174,15 @@ echo "$$" >> "$FM_HOME/state/arm-ran" printf 'watcher: started pid=%s (beacon fresh)\n' "$$" printf 'stale: fixture-win actionable\n' exit 0 +SH + ;; + records-grace) + cat > "$dir/bin/fm-watch-arm.sh" <<'SH' +#!/usr/bin/env bash +echo "$$" >> "$FM_HOME/state/arm-ran" +printf '%s\n' "${FM_GUARD_GRACE:-unset}" > "$FM_HOME/state/arm-received-grace" +printf 'watcher: attached pid=%s (beacon 2s)\n' "$$" +exit 0 SH ;; *) @@ -1156,6 +1165,18 @@ test_active_in_marked_secondmate_home() { pass "auto-arm: active in a marked secondmate home" } +test_long_poll_grace_reaches_arm_wrapper() { + local dir out status + dir=$(make_primary_dir "$TMP_ROOT/long-poll-grace") + : > "$dir/state/task.meta" + write_arm_fixture "$dir" records-grace + out=$(unset FM_GUARD_GRACE; FM_POLL=900 run_autoarm "$dir" 2>/dev/null); status=$? + expect_code 2 "$status" "an unverified close without a healthy watcher must still fail closed" + [ -e "$dir/state/arm-received-grace" ] || fail "arm wrapper never recorded FM_GUARD_GRACE" + [ "$(cat "$dir/state/arm-received-grace")" = 960 ] || fail "arm wrapper must see the poll-derived grace (900+60), got: $(cat "$dir/state/arm-received-grace")" + pass "auto-arm: a long FM_POLL with FM_GUARD_GRACE unset reaches fm-watch-arm.sh with the derived grace" +} + test_fm_lock_status_still_works_with_shared_lib() { local out out=$(FM_HOME="$TMP_ROOT/lock-status-home" bash "$ROOT/bin/fm-lock.sh" status 2>&1) @@ -1202,4 +1223,5 @@ test_superseded_owner_goes_silent_and_never_double_translates test_need_vanished_mid_cycle_closes_quietly test_afk_mid_cycle_suppresses_rewake test_active_in_marked_secondmate_home +test_long_poll_grace_reaches_arm_wrapper test_fm_lock_status_still_works_with_shared_lib diff --git a/tests/fm-control-relaunch.test.sh b/tests/fm-control-relaunch.test.sh index fe2f81bc5c9..b5c58ba1b1d 100755 --- a/tests/fm-control-relaunch.test.sh +++ b/tests/fm-control-relaunch.test.sh @@ -403,7 +403,7 @@ test_relaunch_serializes_concurrent_durable_metadata_publication() { FM_FAKE_TRACE_RELEASE="$launch_release" \ run_control "$dir" rl28 relaunch --note "continue after publication" > "$dir/control.out" & control_pid=$! - while [ ! -e "$prepare" ] && [ "$i" -lt 200 ]; do + while [ ! -e "$prepare" ] && [ "$i" -lt 500 ]; do /bin/sleep 0.01 i=$((i + 1)) done @@ -422,7 +422,7 @@ test_relaunch_serializes_concurrent_durable_metadata_publication() { --carry-platform x --carry-max 280 > "$dir/link.out" 2>&1 & link_pid=$! i=0 - while [ ! -e "$waiting" ] && [ "$i" -lt 200 ]; do + while [ ! -e "$waiting" ] && [ "$i" -lt 500 ]; do /bin/sleep 0.01 i=$((i + 1)) done @@ -435,7 +435,7 @@ test_relaunch_serializes_concurrent_durable_metadata_publication() { } : > "$launch_release" i=0 - while [ ! -e "$ready" ] && [ "$i" -lt 200 ]; do + while [ ! -e "$ready" ] && [ "$i" -lt 500 ]; do /bin/sleep 0.01 i=$((i + 1)) done diff --git a/tests/fm-pi-branch-extension.test.sh b/tests/fm-pi-branch-extension.test.sh index d74c0704c70..f86749aa53e 100644 --- a/tests/fm-pi-branch-extension.test.sh +++ b/tests/fm-pi-branch-extension.test.sh @@ -106,6 +106,10 @@ export class ModelRuntime { constructor() { this.models = (globalThis.__fmBranchStaticModels?.() ?? []).map((model) => ({ ...model })); this.authenticated = new Set(this.models.filter((model) => model.storedAuth !== false).map((model) => model.provider)); + this.registeredProviderConfigs = new Map(); + // Like the real runtime, a registered provider's credentials are only + // known once refresh() has run for it; registration alone is provisional. + this.pendingAuth = new Set(); } static async create() { const queuedError = globalThis.__fmModelRuntimeErrors?.shift(); @@ -115,6 +119,18 @@ export class ModelRuntime { (globalThis.__fmModelRuntimes ??= []).push(runtime); return runtime; } + registerProvider(providerId, config) { + this.registeredProviderConfigs.set(providerId, config); + for (const model of config.models ?? []) { + this.models.push({ ...model, provider: providerId }); + } + if (config.oauth || config.apiKey) this.pendingAuth.add(providerId); + } + async refresh(options) { + for (const providerId of options?.providers ?? this.pendingAuth) { + if (this.pendingAuth.delete(providerId)) this.authenticated.add(providerId); + } + } getModel(provider, id) { return this.models.find((model) => model.provider === provider && model.id === id); } @@ -446,6 +462,8 @@ const modelRegistry = { getAvailable: () => registryModels.filter((model) => model.mainAvailable !== false).slice(), find: (provider, id) => registryModels.find((model) => model.provider === provider && model.id === id), hasConfiguredAuth: (model) => model.mainAvailable !== false, + getRegisteredProviderConfig: (providerId) => globalThis.__fmExtensionProviderConfigs?.get(providerId), + getRegisteredProviderIds: () => [...(globalThis.__fmExtensionProviderConfigs?.keys() ?? [])], }; function makeCtx(extra) { return { @@ -4735,6 +4753,98 @@ EOF pass "a failed cursor write re-delivers a routine note exactly once more while a captain outcome stays deduplicated" } +test_extension_registered_provider_resolves_in_the_branch() { + local repo home out status + repo="$TMP_ROOT/extprov-root" + home="$TMP_ROOT/extprov-home" + mkdir -p "$home/state" "$home/config" + install_pi_branch_extension_fixture "$repo" + PLUGIN="$repo/.pi/extensions/fm-branch-supervision.ts" FM_HOME="$home" FM_ROOT_OVERRIDE="$ROOT" \ + DRIVER_PRELUDE="$DRIVER_PRELUDE" node --input-type=module > "$TMP_ROOT/node-output" 2>&1 <<'EOF' +const prelude = process.env.DRIVER_PRELUDE; +await eval(`(async () => { ${prelude}; globalThis.__t = { fire, dispatch, settle, makeCtx, registryModels, uiSelections, uiPrompts, notices, commands, home }; })()`); +const { fire, dispatch, settle, makeCtx, registryModels, uiSelections, uiPrompts, notices, commands, home } = globalThis.__t; +import { readFileSync, writeFileSync } from "node:fs"; + +// An extension-registered provider exists only in main's registry, never in +// the isolated branch runtime's static catalog. Registering its config on +// main's registry is what makes it resolvable for the branch. +registryModels.push( + { provider: "anthropic", id: "main-model" }, + // Available in main's registry but absent from the branch runtime's static + // catalog, exactly like a provider an extension registered at runtime. + { provider: "devin", id: "swe-1-7", branchAvailable: false }, +); +globalThis.__fmExtensionProviderConfigs = new Map([ + [ + "devin", + { + name: "Devin (Cognition)", + api: "devin-cloud", + baseUrl: "https://server.codeium.com", + models: [{ id: "swe-1-7", name: "SWE 1.7", reasoning: false, input: ["text"], cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200000, maxTokens: 8192 }], + oauth: { name: "Devin (Cognition / Windsurf)", login: async () => ({}), refreshToken: async (c) => c, getApiKey: (c) => c.access }, + streamSimple: () => {}, + }, + ], +]); + +await fire("session_start", {}, makeCtx()); + +// The picker must offer the extension-registered model: it is available in +// main's registry and resolvable in the branch runtime once its registration +// is copied across. +const command = commands.get("supervision-model"); +if (!command) throw new Error("the supervision-model command was not registered"); +uiSelections.push("devin/swe-1-7"); +await command.handler("", makeCtx()); +const offered = uiPrompts[0]; +if (!offered.options.includes("devin/swe-1-7")) { + throw new Error(`the picker must offer an extension-registered provider the branch can run: ${JSON.stringify(offered.options)}`); +} +if (readFileSync(`${home}/config/supervision-branch-model`, "utf8") !== "devin/swe-1-7\n") { + throw new Error("the extension-registered pick was not persisted"); +} +dispatch("signal: extension provider pin"); +await settle(() => (globalThis.__fmSessions ?? []).length === 1, "pinned extension-provider branch build"); +const pinned = globalThis.__fmSessions[0].options.model; +if (!pinned || pinned.provider !== "devin" || pinned.id !== "swe-1-7") { + throw new Error(`the extension-registered pin did not bind the branch: ${JSON.stringify(pinned)}`); +} +// Copying the provider registration must not loosen the branch's isolation: +// the devin-pinned session still loads no extensions, skills, or context files. +const pinnedLoader = globalThis.__fmLoaders.at(-1); +for (const key of ["noExtensions", "noSkills", "noContextFiles"]) { + if (pinnedLoader.options[key] !== true) throw new Error(`devin-pinned branch loader must keep ${key}`); +} + +// Without the registration, the same pin is unavailable and the branch +// refuses to build rather than silently downgrading. +globalThis.__fmExtensionProviderConfigs = new Map(); +await fire("session_shutdown", {}); +await fire("session_start", {}, makeCtx()); +const unregisteredOffer = dispatch("signal: unregistered provider pin"); +if (!unregisteredOffer.accepted) throw new Error("unregistered-pin wake was not initially accepted"); +const unregisteredFailure = await unregisteredOffer.settlement.then( + () => null, + (error) => error, +); +if ( + !(unregisteredFailure instanceof Error) || + !unregisteredFailure.message.includes("devin/swe-1-7") || + !unregisteredFailure.message.includes("supervision model pin") +) { + throw new Error(`the unregistered pin did not reject with its own name: ${String(unregisteredFailure)}`); +} +if ((globalThis.__fmSessions ?? []).length !== 1) throw new Error("an unregistered pin must not build a second branch session"); +process.exit(0); +EOF + status=$? + out=$(cat "$TMP_ROOT/node-output") + expect_code 0 "$status" "an extension-registered provider must resolve in the isolated branch runtime: $out" + pass "an extension-registered provider resolves in the isolated branch runtime" +} + test_outcomes_tool_uses_stock_execution_and_export_consumers test_real_pi_picker_primitives_stay_bounded_and_searchable test_branch_dispatch_two_stage_filter_and_prefix_contract @@ -4763,6 +4873,7 @@ test_supervision_model_picker_is_bounded_searchable_and_branch_only test_branch_model_picker_keeps_follow_main_first_under_ranking test_branch_effort_pin_applies_and_absent_pin_follows_main test_unpinned_branch_follows_main_effort_changes_live +test_extension_registered_provider_resolves_in_the_branch test_supervision_model_command_picks_effort_after_the_model test_unusable_model_pin_falls_back_to_main test_replacement_activation_cleans_leases_and_retries_failure diff --git a/tests/fm-pr-check-security.test.sh b/tests/fm-pr-check-security.test.sh index e5e3eb09ede..07d7a3c79e9 100755 --- a/tests/fm-pr-check-security.test.sh +++ b/tests/fm-pr-check-security.test.sh @@ -964,6 +964,12 @@ test_postrename_poll_validation_revokes_and_retries() { local artifact action dir state destination link_target gate for artifact in data registration check; do for action in type mode device content; do + # The device fault is injected by a fake stat on PATH; on Darwin the + # device helper now calls /usr/bin/stat directly, so the fake can never + # fire there. Skip the device action on Darwin. + if [ "$action" = device ] && [ "$(uname)" = Darwin ]; then + continue + fi dir=$(make_case "poll-final-$artifact-$action") state="$dir/home/state" write_poll_meta "$state" task-a https://github.com/o/r/pull/1 diff --git a/tests/fm-procevent-when.test.sh b/tests/fm-procevent-when.test.sh index 259286beb85..396c48df8d9 100755 --- a/tests/fm-procevent-when.test.sh +++ b/tests/fm-procevent-when.test.sh @@ -21,23 +21,10 @@ export FM_PROCEVENT_CLAIM_ROOT="$TMP_ROOT/claims" pe() { FM_HOME="$1" "$ROOT/bin/fm-procevent.sh" "${@:2}"; } when() { FM_HOME="$1" "$ROOT/bin/fm-procevent-when.sh" "${@:2}"; } -# Every home this suite arms is tracked so teardown can stop any runner still -# blocked on a condition that never fires. -WHEN_HOMES=() -when_teardown() { - local home seen=$'\n' - for home in ${WHEN_HOMES[@]+"${WHEN_HOMES[@]}"}; do - case "$seen" in - *$'\n'"$home"$'\n'*) continue ;; - esac - seen+="$home"$'\n' - FM_HOME="$home" "$ROOT/bin/fm-procevent.sh" sweep-home >/dev/null 2>&1 || true - done - fm_test_cleanup -} -trap when_teardown EXIT - -new_home() { mkdir -p "$1/state"; WHEN_HOMES+=("$1"); } +# Every home this suite arms is registered with tests/lib.sh, which sweeps it +# from every cleanup path so a runner still blocked on a condition that never +# fires cannot survive the run. +new_home() { mkdir -p "$1/state"; fm_test_track_procevent_home "$1"; } wake_payloads() { awk -F '\t' '{print $5}' "$1/state/.wake-queue" 2>/dev/null; } diff --git a/tests/fm-procevent.test.sh b/tests/fm-procevent.test.sh index 7d1fca26166..a7e7c474681 100755 --- a/tests/fm-procevent.test.sh +++ b/tests/fm-procevent.test.sh @@ -24,9 +24,13 @@ BLOCKER="$TMP_ROOT/blocker.sh" cat > "$BLOCKER" <<'SH' #!/usr/bin/env bash # Blocks until the trigger exists, then emits its payload. Completion is the -# event; nothing here polls on a schedule. +# event; nothing here polls on a schedule. The wait is bounded so a stub that +# escapes its test cannot keep spawning processes indefinitely. trigger=$1; shift -while [ ! -e "$trigger" ]; do sleep 0.05; done +while [ ! -e "$trigger" ]; do + [ "$SECONDS" -lt "${FM_TEST_STUB_MAX_BLOCK_SECONDS:-120}" ] || exit 75 + sleep 0.05 +done [ -n "${BLOCKER_STDERR:-}" ] && printf 'noise on stderr\n' >&2 [ -n "${BLOCKER_EXIT:-}" ] && exit "$BLOCKER_EXIT" printf '%s\n' "$@" @@ -35,31 +39,17 @@ chmod +x "$BLOCKER" pe() { FM_HOME="$1" "$ROOT/bin/fm-procevent.sh" "${@:2}"; } -# Every source this suite registers is tracked so teardown can stop its runner. -# A runner started by reconcile is detached and reparented, so a source that -# never completes outlives the suite unless it is retired explicitly - removing -# the fixture directory does not stop an already-running child. -PE_TRACKED=() +# Every home this suite registers a source in is tracked so teardown can stop +# its runners. A runner started by reconcile is detached and reparented, so a +# source that never completes outlives the suite unless its home is swept - +# removing the fixture directory does not stop an already-running child. +# tests/lib.sh owns that sweep and runs it from every cleanup path. pe_register() { # -- ... local home=$1 adapter=$2 id=$3 shift 3 - PE_TRACKED+=("$home|$id") + fm_test_track_procevent_home "$home" pe "$home" register "$adapter" "$id" "$@" } - -procevent_teardown() { - local entry home seen=$'\n' - for entry in ${PE_TRACKED[@]+"${PE_TRACKED[@]}"}; do - home=${entry%%|*} - case "$seen" in - *$'\n'"$home"$'\n'*) continue ;; - esac - seen+="$home"$'\n' - FM_HOME="$home" "$ROOT/bin/fm-procevent.sh" sweep-home >/dev/null 2>&1 || true - done - fm_test_cleanup -} -trap procevent_teardown EXIT new_home() { mkdir -p "$1/state"; } wake_payloads() { awk -F '\t' '{print $5}' "$1/state/.wake-queue" 2>/dev/null; } @@ -417,7 +407,7 @@ pe_adapter() { # ...: run the runner against the fixture adapte } HPUBLISH="$TMP_ROOT/hpublish"; new_home "$HPUBLISH" -PE_TRACKED+=("$HPUBLISH|publish-src") +fm_test_track_procevent_home "$HPUBLISH" pe_adapter "$HPUBLISH" register applying publish-src -- /bin/echo "apply after publish" >/dev/null mkdir "$HPUBLISH/state/.wake-queue" out=$(pe_adapter "$HPUBLISH" start publish-src 2>&1) @@ -449,7 +439,7 @@ pass "automatic application waits for durable publication and failed publication # channel is the announcement. The declaration never silences a capture the # adapter could NOT apply - that one still publishes for the handler. HSELF="$TMP_ROOT/hself"; new_home "$HSELF" -PE_TRACKED+=("$HSELF|self-src") +fm_test_track_procevent_home "$HSELF" pe_adapter "$HSELF" register selfann self-src -- /bin/echo "self announced" >/dev/null out=$(pe_adapter "$HSELF" start self-src 2>&1) assert_contains "$out" "autohandled: self-src" "the self-announcing adapter did not apply its own capture" @@ -480,7 +470,7 @@ rm -f "$HSELF/state/selfann-fail" pass "a self-announcing adapter applies quietly and still publishes what it could not apply" HTERM="$TMP_ROOT/hterm"; new_home "$HTERM" -PE_TRACKED+=("$HTERM|ends-src") +fm_test_track_procevent_home "$HTERM" pe_adapter "$HTERM" register endnow ends-src -- /bin/echo "terminal payload" >/dev/null out=$(pe_adapter "$HTERM" start ends-src) assert_contains "$out" "captured:" "a terminal result is still captured durably" @@ -504,7 +494,7 @@ assert_contains "$out" "published=0" "an acknowledged terminal result stops bein pass "an adapter-classified terminal result is captured once, announced, and retires its source automatically" HOPEN="$TMP_ROOT/hopen"; new_home "$HOPEN" -PE_TRACKED+=("$HOPEN|open-src") +fm_test_track_procevent_home "$HOPEN" pe_adapter "$HOPEN" register openended open-src -- /bin/echo "open payload" >/dev/null out=$(pe_adapter "$HOPEN" start open-src) assert_contains "$out" "captured:" "a result from an adapter with no terminal verdict is captured" @@ -514,12 +504,22 @@ pe_adapter "$HOPEN" retire open-src >/dev/null pass "a source stays armed unless its own adapter classifies the result terminal" HREPLACE="$TMP_ROOT/hreplace"; new_home "$HREPLACE" -PE_TRACKED+=("$HREPLACE|replace-src") +fm_test_track_procevent_home "$HREPLACE" OLD_TRIGGER="$TMP_ROOT/replace-old-trigger" -pe_adapter "$HREPLACE" register endnow replace-src -- "$BLOCKER" "$OLD_TRIGGER" "old terminal payload" >/dev/null +OLD_STARTED="$TMP_ROOT/replace-old-started" +REPLACE_BLOCKER="$TMP_ROOT/replace-blocker.sh" +cat > "$REPLACE_BLOCKER" <<'SH' +#!/usr/bin/env bash +printf 'started\n' > "$1" +shift +exec "$@" +SH +chmod +x "$REPLACE_BLOCKER" +pe_adapter "$HREPLACE" register endnow replace-src -- \ + "$REPLACE_BLOCKER" "$OLD_STARTED" "$BLOCKER" "$OLD_TRIGGER" "old terminal payload" >/dev/null pe_adapter "$HREPLACE" start replace-src > "$TMP_ROOT/replace-old.out" 2>&1 & replace_old_pid=$! -wait_for "$FM_PROCEVENT_CLAIM_ROOT/replace-src.claim" || fail "the old registration was never claimed" +wait_for "$OLD_STARTED" || fail "the old registration never started" pe_adapter "$HREPLACE" register openended replace-src -- /bin/echo "replacement payload" >/dev/null touch "$OLD_TRIGGER" wait "$replace_old_pid" || fail "the old terminal runner failed" @@ -537,7 +537,7 @@ pe_adapter "$HREPLACE" retire replace-src >/dev/null pass "terminal retirement preserves and releases a concurrently replaced registration" HRETFAIL="$TMP_ROOT/hretfail"; new_home "$HRETFAIL" -PE_TRACKED+=("$HRETFAIL|retire-fail-src") +fm_test_track_procevent_home "$HRETFAIL" FAIL_RM_BIN=$(fm_fakebin "$TMP_ROOT/retire-fail-bin") REAL_RM=$(command -v rm) export REAL_RM @@ -598,7 +598,7 @@ chmod +x "$LAVISH_BIN/lavish-axi" REVIEW_ART="$TMP_ROOT/review.html" printf '

review

\n' > "$REVIEW_ART" lavish_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$REVIEW_ART") -PE_TRACKED+=("$HLT|$lavish_id") +fm_test_track_procevent_home "$HLT" PATH="$LAVISH_BIN:$PATH" FM_HOME="$HLT" "$ROOT/bin/fm-procevent-lavish.sh" arm "$REVIEW_ART" >/dev/null for _ in $(seq 1 6); do PATH="$LAVISH_BIN:$PATH" pe "$HLT" reconcile >/dev/null @@ -638,7 +638,7 @@ chmod +x "$EMPTY_BIN/lavish-axi" QUIET_ART="$TMP_ROOT/quiet-board.html" printf '

quiet

\n' > "$QUIET_ART" quiet_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$QUIET_ART") -PE_TRACKED+=("$HEMPTY|$quiet_id") +fm_test_track_procevent_home "$HEMPTY" PATH="$EMPTY_BIN:$PATH" FM_HOME="$HEMPTY" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$QUIET_ART" >/dev/null quiet_out=$(PATH="$EMPTY_BIN:$PATH" pe "$HEMPTY" start "$quiet_id" 2>&1) @@ -684,7 +684,7 @@ chmod +x "$ANSWER_BIN/lavish-axi" ANSWER_ART="$TMP_ROOT/answered-board.html" printf '

answered

\n' > "$ANSWER_ART" answer_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$ANSWER_ART") -PE_TRACKED+=("$HANSWER|$answer_id") +fm_test_track_procevent_home "$HANSWER" PATH="$ANSWER_BIN:$PATH" FM_HOME="$HANSWER" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$ANSWER_ART" >/dev/null PATH="$ANSWER_BIN:$PATH" pe "$HANSWER" reconcile >/dev/null @@ -736,9 +736,27 @@ esac SH chmod +x "$LAVISH_SCRIPTED_BIN/lavish-axi" export LAVISH_COUNT LAVISH_SCRIPT + +DEFAULT_RATE_ART="$TMP_ROOT/default-rate-board.html" +printf '

default rate

\n' > "$DEFAULT_RATE_ART" +DEFAULT_RATE_COUNT="$TMP_ROOT/default-rate-count" +PATH="$LAVISH_SCRIPTED_BIN:$PATH" LAVISH_COUNT="$DEFAULT_RATE_COUNT" LAVISH_SCRIPT=interrupt \ + FM_LAVISH_POLL_RETRY_DELAY='' \ + "$ROOT/bin/fm-procevent-lavish.sh" poll "$DEFAULT_RATE_ART" >/dev/null 2>&1 & +DEFAULT_RATE_PID=$! +perl -MTime::HiRes=sleep -e 'sleep 6.2' +kill -TERM "$DEFAULT_RATE_PID" 2>/dev/null || true +wait "$DEFAULT_RATE_PID" 2>/dev/null || true +default_rate_count=$(cat "$DEFAULT_RATE_COUNT" 2>/dev/null || echo 0) +[ "$default_rate_count" -ge 2 ] \ + || fail "the default poll governor stopped an instantly returning source from making progress" +[ "$default_rate_count" -le 2 ] \ + || fail "the shipped poll governor allowed $default_rate_count iterations in 6.2 seconds" +pass "the shipped poll governor bounds an instantly returning source" + # A bounded test override keeps the retry policy's real bound under test without # making the suite wait out the production delay. -export FM_LAVISH_POLL_RETRY_DELAY=0 +export FM_LAVISH_POLL_RETRY_DELAY=1 # Two interruptions, then the captain's real feedback: the retries are silent and # only the feedback becomes a captured result and a check wake. @@ -746,7 +764,7 @@ HRETRY="$TMP_ROOT/hretry"; new_home "$HRETRY" RETRY_ART="$TMP_ROOT/retry-board.html" printf '

retry

\n' > "$RETRY_ART" retry_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$RETRY_ART") -PE_TRACKED+=("$HRETRY|$retry_id") +fm_test_track_procevent_home "$HRETRY" LAVISH_COUNT="$TMP_ROOT/retry-count"; LAVISH_SCRIPT="interrupt interrupt feedback" PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HRETRY" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$RETRY_ART" >/dev/null @@ -770,7 +788,7 @@ HEXH="$TMP_ROOT/hexh"; new_home "$HEXH" EXH_ART="$TMP_ROOT/exhaust-board.html" printf '

exhaust

\n' > "$EXH_ART" exh_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$EXH_ART") -PE_TRACKED+=("$HEXH|$exh_id") +fm_test_track_procevent_home "$HEXH" LAVISH_COUNT="$TMP_ROOT/exhaust-count"; LAVISH_SCRIPT="interrupt" PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HEXH" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$EXH_ART" >/dev/null @@ -793,7 +811,7 @@ HOTHER="$TMP_ROOT/hother"; new_home "$HOTHER" OTHER_ART="$TMP_ROOT/other-board.html" printf '

other

\n' > "$OTHER_ART" other_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$OTHER_ART") -PE_TRACKED+=("$HOTHER|$other_id") +fm_test_track_procevent_home "$HOTHER" LAVISH_COUNT="$TMP_ROOT/other-count"; LAVISH_SCRIPT="other-server-error" PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HOTHER" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$OTHER_ART" >/dev/null @@ -813,9 +831,9 @@ HNEAR="$TMP_ROOT/hnear"; new_home "$HNEAR" NEAR_ART="$TMP_ROOT/near-board.html" printf '

near

\n' > "$NEAR_ART" near_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$NEAR_ART") -PE_TRACKED+=("$HNEAR|$near_id") +fm_test_track_procevent_home "$HNEAR" LAVISH_COUNT="$TMP_ROOT/near-count"; LAVISH_SCRIPT="near-interrupt feedback" -PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HNEAR" FM_LAVISH_POLL_RETRY_DELAY=0 \ +PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HNEAR" FM_LAVISH_POLL_RETRY_DELAY=1 \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$NEAR_ART" >/dev/null PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HNEAR" pe "$HNEAR" start "$near_id" >/dev/null [ "$(cat "$LAVISH_COUNT")" = 1 ] \ @@ -832,14 +850,14 @@ HINVALID="$TMP_ROOT/hinvalid"; new_home "$HINVALID" INVALID_ART="$TMP_ROOT/invalid-delay-board.html" printf '

invalid delay

\n' > "$INVALID_ART" invalid_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$INVALID_ART") -for invalid_delay in 61 invalid; do +for invalid_delay in 0 61 invalid; do invalid_status=0 invalid_out=$(PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HINVALID" \ FM_LAVISH_POLL_RETRY_DELAY="$invalid_delay" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$INVALID_ART" 2>&1) || invalid_status=$? [ "$invalid_status" -ne 0 ] \ || fail "arm accepted invalid retry delay: $invalid_delay" - assert_contains "$invalid_out" "must be whole seconds from 0 to 60" \ + assert_contains "$invalid_out" "must be whole seconds from 1 to 60" \ "arm explains the rejected retry delay" assert_absent "$HINVALID/state/procevent/$invalid_id.source" \ "arm publishes no source registration for an invalid retry delay" @@ -865,7 +883,7 @@ LAVISH_STREAM_RELEASE="$TMP_ROOT/stream-release" mkdir -p "$STREAM_TMPDIR" printf '

stream

\n' > "$STREAM_ART" stream_id=$("$ROOT/bin/fm-procevent-lavish.sh" source-id "$STREAM_ART") -PE_TRACKED+=("$HSTREAM|$stream_id") +fm_test_track_procevent_home "$HSTREAM" LAVISH_COUNT="$TMP_ROOT/stream-count"; LAVISH_SCRIPT="stream" PATH="$LAVISH_SCRIPTED_BIN:$PATH" FM_HOME="$HSTREAM" \ "$ROOT/bin/fm-procevent-lavish.sh" arm "$STREAM_ART" >/dev/null @@ -1108,32 +1126,19 @@ kill -0 -"$orphan_leader" 2>/dev/null || fail "fixture invalid: the owned child orphan_out=$(pe "$HG" reconcile) kill -0 -"$orphan_leader" 2>/dev/null \ - && fail "reconcile left the crashed generation's process group alive: $orphan_out" -sleep 0.5 -assert_absent "$ORPHAN_OVERLAP" "no replacement source starts while the crashed generation remains alive" -case "$orphan_out" in - *"started=1"*) - # The replacement is detached: it records its own claim and execs its source - # after reconcile has already returned, so both effects must be waited for - # rather than snapshotted behind the settle window above. - wait_for "$FM_PROCEVENT_CLAIM_ROOT/orphan-src.claim" \ - || fail "a replacement runner started without recording its own claim" - wait_for_lines "$ORPHAN_LOG" 2 \ - || fail "the replacement runner never started its source: $(cat "$ORPHAN_LOG")" - [ "$(wc -l < "$ORPHAN_LOG" | tr -d ' ')" = 2 ] \ - || fail "reconcile did not start exactly one replacement source: $(cat "$ORPHAN_LOG")" - ;; - *"started=0"*) - [ -e "$FM_PROCEVENT_CLAIM_ROOT/orphan-src.claim" ] \ - || fail "refusing to replace must preserve the claim for retry: $orphan_out" - [ "$(wc -l < "$ORPHAN_LOG" | tr -d ' ')" = 1 ] \ - || fail "reconcile started a source while refusing replacement: $(cat "$ORPHAN_LOG")" - ;; - *) fail "unexpected reconcile result for a crashed leader: $orphan_out" ;; -esac -: > "$ORPHAN_TRIGGER" + || fail "reconcile signalled an ambiguous leaderless process group: $orphan_out" +assert_contains "$orphan_out" "started=0" \ + "reconcile does not replace an ambiguous leaderless generation" +[ -e "$FM_PROCEVENT_CLAIM_ROOT/orphan-src.claim" ] \ + || fail "refusing ambiguous cleanup must preserve the claim" +[ "$(wc -l < "$ORPHAN_LOG" | tr -d ' ')" = 1 ] \ + || fail "reconcile started a source beside an ambiguous leaderless group" +assert_absent "$ORPHAN_OVERLAP" "no replacement source starts while the leaderless group remains" +kill -KILL -"$orphan_leader" 2>/dev/null || true +for _ in $(seq 1 50); do kill -0 -"$orphan_leader" 2>/dev/null || break; sleep 0.1; done +kill -0 -"$orphan_leader" 2>/dev/null && fail "could not clean up the leaderless fixture group" pe "$HG" retire orphan-src >/dev/null -pass "a crashed runner leader never lets a live owned group be reclaimed as stale" +pass "an ambiguous leaderless group is preserved without replacement" # Counterexample: a genuinely dead generation - no leader and no surviving # group - must still be reclaimable, or crash recovery would deadlock. @@ -1350,7 +1355,7 @@ kill -0 "$innocent_pid" 2>/dev/null || fail "retirement signaled a PID whose ide kill "$innocent_pid" 2>/dev/null || true wait "$innocent_pid" 2>/dev/null || true assert_absent "$FM_PROCEVENT_CLAIM_ROOT/reused-src.claim" "retirement releases the exact reused-pid claim" -pass "PID reuse cannot signal an unrelated process" +pass "detected PID reuse is refused before signalling" HL="$TMP_ROOT/hl"; new_home "$HL" IDENTITY_TRIGGER="$TMP_ROOT/identity-trigger" @@ -1385,6 +1390,19 @@ wait_for "$FM_PROCEVENT_CLAIM_ROOT/sweep-one.claim" || fail "home sweep fixture wait_for "$FM_PROCEVENT_CLAIM_ROOT/sweep-two.claim" || fail "home sweep fixture two did not start" sweep_pid_one=$(sed -n '2p' "$FM_PROCEVENT_CLAIM_ROOT/sweep-one.claim") sweep_pid_two=$(sed -n '2p' "$FM_PROCEVENT_CLAIM_ROOT/sweep-two.claim") +# The claim-only case is an owned claim with no live runner, so build exactly +# that: kill the runner's group so it cannot run its own cleanup, confirm it is +# gone, and only then drop the registration. Deleting the registration out from +# under a LIVE runner no longer produces this case, because a superseded +# generation now observes the identity mismatch, self-retires, and releases its +# claim - so the sweep would race that exit and see one source or two depending +# on which won. +kill -KILL -"$sweep_pid_two" 2>/dev/null || true +for _ in $(seq 1 50); do kill -0 "$sweep_pid_two" 2>/dev/null || break; sleep 0.1; done +kill -0 "$sweep_pid_two" 2>/dev/null \ + && fail "the claim-only sweep fixture runner did not stop" +assert_present "$FM_PROCEVENT_CLAIM_ROOT/sweep-two.claim" \ + "a killed runner leaves its owned claim behind for the sweep" rm -f "$HM/state/procevent/sweep-two.source" out=$(pe "$HM" sweep-home --preflight) assert_contains "$out" "sweep preflight: ready" "home sweep preflight validates the full bounded snapshot" @@ -1512,6 +1530,64 @@ kill -0 "$noisy_child" 2>/dev/null && fail "TERM-resistant source child survived assert_absent "$staged" "retirement removes the tracked partial staging file" pass "live output stays bounded and retirement reaps the whole source group" +HPOST_TERM="$TMP_ROOT/post-term-reuse"; new_home "$HPOST_TERM" +POST_TERM_SOURCE="$TMP_ROOT/post-term-reuse-source.sh" +POST_TERM_PID="$TMP_ROOT/post-term-reuse.pid" +POST_TERM_MARKER="$TMP_ROOT/post-term-reuse.marker" +POST_TERM_COUNT="$TMP_ROOT/post-term-reuse.count" +cat > "$POST_TERM_SOURCE" <<'SH' +#!/usr/bin/env bash +trap '' TERM +printf '%s\n' "$$" > "$1" +while :; do sleep 1; done +SH +chmod +x "$POST_TERM_SOURCE" +POST_TERM_BIN=$(fm_fakebin "$TMP_ROOT/post-term-reuse-bin") +REAL_PS=$(command -v ps) || fail "the post-TERM reuse fixture requires ps" +cat > "$POST_TERM_BIN/ps" < "$POST_TERM_COUNT" + if [ "\$count" -gt 1 ]; then + printf 'post-TERM reused identity\n' + exit 0 + fi +fi +exec "$REAL_PS" "\$@" +SH +chmod +x "$POST_TERM_BIN/ps" +pe_register "$HPOST_TERM" lavish post-term-src -- \ + "$POST_TERM_SOURCE" "$POST_TERM_PID" >/dev/null +FM_PROCEVENT_OWNER_CHECK_SECONDS=5 pe "$HPOST_TERM" reconcile >/dev/null +wait_for "$POST_TERM_PID" || fail "the post-TERM reuse fixture did not start" +wait_for "$FM_PROCEVENT_CLAIM_ROOT/post-term-src.claim" \ + || fail "the post-TERM reuse fixture did not claim its source" +POST_TERM_RUNNER=$(sed -n '2p' "$FM_PROCEVENT_CLAIM_ROOT/post-term-src.claim") +touch "$POST_TERM_MARKER" +post_term_status=0 +PATH="$POST_TERM_BIN:$PATH" FM_PROC_ROOT_OVERRIDE="$TMP_ROOT/no-post-term-proc" \ + pe "$HPOST_TERM" retire post-term-src >/dev/null 2>&1 || post_term_status=$? +[ "$post_term_status" -ne 0 ] || fail "retirement escalated after runner identity became ambiguous" +# Which ambiguity the escalation meets here is platform-dependent, so this +# asserts the invariant both forms share rather than one form's internals. +# Where the runner leader keeps waiting on its TERM-ignoring source child the +# post-TERM check sees a live leader whose identity no longer matches, and +# where the leader dies promptly it sees a leaderless group carrying the same +# numeric id; fm_procevent_pid_state reaches the second verdict without +# consulting process identity at all, so counting identity lookups pins a +# timing- and platform-dependent internal rather than the behavior. +kill -0 -"$POST_TERM_RUNNER" 2>/dev/null \ + || fail "an ambiguous reused-PID group was killed during escalation" +kill -KILL -"$POST_TERM_RUNNER" 2>/dev/null || true +for _ in $(seq 1 50); do kill -0 -"$POST_TERM_RUNNER" 2>/dev/null || break; sleep 0.1; done +kill -0 -"$POST_TERM_RUNNER" 2>/dev/null && fail "could not clean up the post-TERM fixture group" +pe "$HPOST_TERM" retire post-term-src >/dev/null +pass "cleanup aborts escalation after runner identity becomes ambiguous" + HBAD="$TMP_ROOT/hbad"; new_home "$HBAD" pe_register "$HBAD" lavish bad-limit -- /bin/true >/dev/null bad_limit_status=0 @@ -1901,4 +1977,527 @@ assert_not_contains "$runner_help" "exactly-once" \ "the runner's help claims no exactly-once delivery" pass "the published interfaces state the loss limitation and claim no lossless delivery" +# --- launch pacing and guard startup ---------------------------------------- + +FAST_SOURCE="$TMP_ROOT/fast-source.sh" +cat > "$FAST_SOURCE" <<'SH' +#!/usr/bin/env bash +perl -MTime::HiRes=time -e 'printf "%.6f\n", time' >> "$1" +exit 1 +SH +chmod +x "$FAST_SOURCE" + +STORM_SOURCE="$TMP_ROOT/storm-source.sh" +cat > "$STORM_SOURCE" <<'SH' +#!/usr/bin/env bash +perl -MTime::HiRes=time -e 'printf "%.6f\n", time' >> "$1" +FM_HOME="$2" perl -MPOSIX=setsid -e ' + my @command = @ARGV; + defined(my $pid = fork) or exit 1; + exit 0 if $pid; + setsid() >= 0 or exit 1; + open STDIN, "<", "/dev/null" or exit 1; + open STDOUT, ">", "/dev/null" or exit 1; + open STDERR, ">", "/dev/null" or exit 1; + select undef, undef, undef, 0.2; + exec @command; +' "$3/bin/fm-procevent.sh" reconcile +exit 1 +SH +chmod +x "$STORM_SOURCE" + +HFLOOR="$TMP_ROOT/launch-floor"; new_home "$HFLOOR" +fm_test_track_procevent_home "$HFLOOR" +pe_register "$HFLOOR" lavish floor-src -- \ + "$STORM_SOURCE" "$TMP_ROOT/launch-times" "$HFLOOR" "$ROOT" +FM_PROCEVENT_OWNER_LEASE_SECONDS=4 FM_PROCEVENT_OWNER_CHECK_SECONDS=1 \ + FM_PROCEVENT_LAUNCH_FLOOR_SECONDS=1 pe "$HFLOOR" reconcile >/dev/null +floor_deadline=$((SECONDS + 12)) +while :; do + floor_count=0 + [ ! -f "$TMP_ROOT/launch-times" ] \ + || floor_count=$(wc -l < "$TMP_ROOT/launch-times" | tr -d ' ') + [ "$floor_count" -ge 3 ] && break + [ "$SECONDS" -lt "$floor_deadline" ] \ + || fail "the orphan-storm fixture did not relaunch its source command" + sleep 0.1 +done +launch_count=$(wc -l < "$TMP_ROOT/launch-times" | tr -d ' ') +launch_span=$(perl -e '@t=<>; printf "%.3f", $t[-1] - $t[0]' "$TMP_ROOT/launch-times") +perl -e 'exit($ARGV[0] >= ($ARGV[1] - 1) * 0.8 ? 0 : 1)' "$launch_span" "$launch_count" \ + || fail "an orphaned source launched $launch_count times in only ${launch_span}s" +[ "$launch_count" -le 6 ] \ + || fail "an orphaned source stormed $launch_count launches during its owner-dead grace window" +pass "an orphaned source command obeys the launch floor during its grace window" + +HPACE="$TMP_ROOT/registration-pacing"; new_home "$HPACE" +fm_test_track_procevent_home "$HPACE" +PACE_LOG="$TMP_ROOT/registration-pacing.log" +pe_register "$HPACE" lavish pace-src -- "$FAST_SOURCE" "$PACE_LOG" >/dev/null +FM_PROCEVENT_LAUNCH_FLOOR_SECONDS=3600 pe "$HPACE" start pace-src >/dev/null +pe "$HPACE" retire pace-src >/dev/null +pe_register "$HPACE" lavish pace-src -- "$FAST_SOURCE" "$PACE_LOG" >/dev/null +FM_PROCEVENT_LAUNCH_FLOOR_SECONDS=3600 pe "$HPACE" start pace-src > "$TMP_ROOT/replacement-pacing.out" 2>&1 & +PACE_START_PID=$! +pace_deadline=$((SECONDS + 4)) +while kill -0 "$PACE_START_PID" 2>/dev/null; do + if [ "$SECONDS" -ge "$pace_deadline" ]; then + pe "$HPACE" retire pace-src >/dev/null 2>&1 || true + wait "$PACE_START_PID" 2>/dev/null || true + fail "a replacement registration inherited the prior launch floor" + fi + sleep 0.1 +done +wait "$PACE_START_PID" || fail "the replacement registration failed" +[ "$(wc -l < "$PACE_LOG" | tr -d ' ')" = 2 ] \ + || fail "a replacement registration did not launch immediately" +PACE_STAMPS=$(find "$HPACE/state/procevent" -maxdepth 1 -type f \ + -name 'pace-src.*.last-launch' | wc -l | tr -d ' ') +[ "$PACE_STAMPS" = 1 ] || fail "replacement registrations accumulated stale pacing state" +pass "a replacement registration starts with one fresh launch floor" + +HPACE_RACE="$TMP_ROOT/registration-pacing-race"; new_home "$HPACE_RACE" +fm_test_track_procevent_home "$HPACE_RACE" +PACE_RACE_LOG="$TMP_ROOT/registration-pacing-race.log" +pe_register "$HPACE_RACE" lavish pace-race-src -- "$FAST_SOURCE" "$PACE_RACE_LOG" >/dev/null +FM_PROCEVENT_LAUNCH_FLOOR_SECONDS=3 pe "$HPACE_RACE" start pace-race-src >/dev/null +FM_PROCEVENT_LAUNCH_FLOOR_SECONDS=3 \ + pe "$HPACE_RACE" start pace-race-src > "$TMP_ROOT/registration-pacing-race.out" 2>&1 & +PACE_RACE_PID=$! +wait_for "$FM_PROCEVENT_CLAIM_ROOT/pace-race-src.claim" \ + || fail "the superseded pacing fixture did not claim its registration" +[ "$(wc -l < "$PACE_RACE_LOG" | tr -d ' ')" = 1 ] \ + || fail "the superseded pacing fixture was not waiting on its launch floor" +pe_register "$HPACE_RACE" lavish pace-race-src -- "$FAST_SOURCE" "$PACE_RACE_LOG" >/dev/null +wait "$PACE_RACE_PID" || fail "the superseded paced runner failed" +[ "$(wc -l < "$PACE_RACE_LOG" | tr -d ' ')" = 1 ] \ + || fail "the superseded paced runner invoked its stale command" +# The runner marker is written before the launch floor is waited on, and a home +# sweep counts a marker with no owned claim as a preflight failure. A superseded +# generation that exits without clearing its marker therefore makes the whole +# home refuse to sweep, so assert the marker is gone and the sweep still runs. +assert_absent "$HPACE_RACE/state/procevent/pace-race-src.runner" \ + "a superseded paced runner leaves no runner marker behind" +FM_PROCEVENT_LAUNCH_FLOOR_SECONDS=3 pe "$HPACE_RACE" start pace-race-src >/dev/null +PACE_RACE_STAMPS=$(find "$HPACE_RACE/state/procevent" -maxdepth 1 -type f \ + -name 'pace-race-src.*.last-launch' | wc -l | tr -d ' ') +[ "$PACE_RACE_STAMPS" = 1 ] \ + || fail "a superseded sleeping runner recreated stale pacing state" +pass "a superseded sleeping runner cannot recreate stale pacing state" + +HCOMMIT="$TMP_ROOT/registration-commit"; new_home "$HCOMMIT" +fm_test_track_procevent_home "$HCOMMIT" +COMMIT_LOG="$TMP_ROOT/registration-commit.log" +mkdir -p "$HCOMMIT/state/procevent/commit-src.1-2.last-launch" +pe_register "$HCOMMIT" lavish commit-src -- "$FAST_SOURCE" "$COMMIT_LOG" >/dev/null \ + || fail "post-commit pacing cleanup made registration report failure" +FM_PROCEVENT_LAUNCH_FLOOR_SECONDS=1 pe "$HCOMMIT" start commit-src >/dev/null \ + || fail "a successfully published registration was not executable" +[ "$(wc -l < "$COMMIT_LOG" | tr -d ' ')" = 1 ] \ + || fail "the committed registration did not invoke its source" +pass "post-commit pacing cleanup cannot veto registration publication" + +HROLLBACK="$TMP_ROOT/rollback-pacing"; new_home "$HROLLBACK" +fm_test_track_procevent_home "$HROLLBACK" +ROLLBACK_LOG="$TMP_ROOT/rollback-pacing.log" +pe_register "$HROLLBACK" lavish rollback-src -- "$FAST_SOURCE" "$ROLLBACK_LOG" >/dev/null +FM_PROCEVENT_LAUNCH_FLOOR_SECONDS=1 pe "$HROLLBACK" start rollback-src >/dev/null +ROLLBACK_STAMP= +for candidate in "$HROLLBACK/state/procevent"/rollback-src.*.last-launch; do + [ -f "$candidate" ] && ROLLBACK_STAMP=$candidate +done +[ -n "$ROLLBACK_STAMP" ] || fail "the first launch did not persist its pacing state" +printf '%s\n' "$(( $(date +%s) + 3600 ))" > "$ROLLBACK_STAMP" +FM_PROCEVENT_LAUNCH_FLOOR_SECONDS=3600 pe "$HROLLBACK" start rollback-src > "$TMP_ROOT/rollback.out" 2>&1 & +ROLLBACK_START_PID=$! +rollback_deadline=$((SECONDS + 4)) +while kill -0 "$ROLLBACK_START_PID" 2>/dev/null; do + if [ "$SECONDS" -ge "$rollback_deadline" ]; then + pe "$HROLLBACK" retire rollback-src >/dev/null 2>&1 || true + wait "$ROLLBACK_START_PID" 2>/dev/null || true + fail "a pre-reboot monotonic stamp delayed the first launch" + fi + sleep 0.1 +done +wait "$ROLLBACK_START_PID" || fail "the rollback-paced source failed" +[ "$(wc -l < "$ROLLBACK_LOG" | tr -d ' ')" = 2 ] \ + || fail "the rollback-paced source did not invoke twice" +pass "a pre-reboot monotonic stamp is treated as expired" + +storm_deadline=$((SECONDS + 15)) +while :; do + storm_before=$(wc -l < "$TMP_ROOT/launch-times" | tr -d ' ') + sleep 2 + storm_after=$(wc -l < "$TMP_ROOT/launch-times" | tr -d ' ') + [ "$storm_before" = "$storm_after" ] && break + [ "$SECONDS" -lt "$storm_deadline" ] \ + || fail "an orphaned self-relaunching source survived its expired owner lease" +done +pass "an expired owner lease stops a self-relaunching source generation" + +HRECREATED="$TMP_ROOT/recreated-owner"; new_home "$HRECREATED" +fm_test_track_procevent_home "$HRECREATED" +RECREATED_TRIGGER="$TMP_ROOT/recreated-owner.trigger" +pe_register "$HRECREATED" lavish recreated-src -- \ + "$BLOCKER" "$RECREATED_TRIGGER" "recreated payload" >/dev/null +FM_PROCEVENT_OWNER_LEASE_SECONDS=30 FM_PROCEVENT_OWNER_CHECK_SECONDS=1 \ + pe "$HRECREATED" reconcile >/dev/null +wait_for "$HRECREATED/state/procevent/recreated-src.runner" \ + || fail "the recreated-path fixture never launched its source" +RECREATED_RUNNER_PID=$(cat "$HRECREATED/state/procevent/recreated-src.runner") +mv "$HRECREATED/state" "$TMP_ROOT/recreated-owner-old-state" +pe_register "$HRECREATED" lavish recreated-src -- \ + "$BLOCKER" "$RECREATED_TRIGGER" "replacement payload" >/dev/null +FM_PROCEVENT_OWNER_LEASE_SECONDS=30 FM_PROCEVENT_OWNER_CHECK_SECONDS=1 \ + pe "$HRECREATED" reconcile >/dev/null +recreated_deadline=$((SECONDS + 8)) +while kill -0 "$RECREATED_RUNNER_PID" 2>/dev/null; do + [ "$SECONDS" -lt "$recreated_deadline" ] \ + || fail "a fresh lease at a recreated state path preserved the old runner" + sleep 0.1 +done +pass "a recreated state path does not preserve the old runner" + +HGUARDFAIL="$TMP_ROOT/guard-failure"; new_home "$HGUARDFAIL" +fm_test_track_procevent_home "$HGUARDFAIL" +pe_register "$HGUARDFAIL" lavish guard-fail-src -- "$FAST_SOURCE" "$TMP_ROOT/unguarded-launches" +guard_fail_status=0 +guard_fail_out=$(FM_PROCEVENT_OWNER_LEASE_SECONDS=invalid \ + pe "$HGUARDFAIL" start guard-fail-src 2>&1) || guard_fail_status=$? +[ "$guard_fail_status" -ne 0 ] || fail "a runner continued after its owner guard failed to initialize" +assert_contains "$guard_fail_out" "cannot start the runner's owner guard" \ + "guard initialization failure is reported at the runner boundary" +assert_absent "$TMP_ROOT/unguarded-launches" \ + "a source command ran without a successfully initialized owner guard" +pass "a runner fails closed when its owner guard cannot initialize" + +HATTACHED="$TMP_ROOT/attached-owner"; new_home "$HATTACHED" +fm_test_track_procevent_home "$HATTACHED" +ATTACHED_TRIGGER="$TMP_ROOT/attached.trigger" +pe_register "$HATTACHED" lavish attached-src -- "$BLOCKER" "$ATTACHED_TRIGGER" "attached payload" +FM_PROCEVENT_OWNER_LEASE_SECONDS=1 FM_PROCEVENT_OWNER_CHECK_SECONDS=1 \ + pe "$HATTACHED" start attached-src > "$TMP_ROOT/attached.out" 2>&1 & +ATTACHED_START_PID=$! +wait_for "$HATTACHED/state/procevent/attached-src.runner" \ + || fail "the attached start never launched its source" +sleep 4 +kill -0 "$ATTACHED_START_PID" 2>/dev/null \ + || fail "a foreground start lost its owner lease while its caller remained attached" +touch "$ATTACHED_TRIGGER" +wait "$ATTACHED_START_PID" || fail "the attached start did not complete after its source returned" +assert_contains "$(cat "$TMP_ROOT/attached.out")" "captured:" \ + "the attached source result was not captured" +pass "a foreground start refreshes its lease while its caller remains attached" + +HCLOCK="$TMP_ROOT/lease-clock"; new_home "$HCLOCK" +fm_test_track_procevent_home "$HCLOCK" +CLOCK_TRIGGER="$TMP_ROOT/lease-clock.trigger" +CLOCK_STATE="$TMP_ROOT/lease-clock-state" +CLOCK_BIN=$(fm_fakebin "$TMP_ROOT/lease-clock-bin") +REAL_DATE=$(command -v date) || fail "the lease clock fixture requires date" +cat > "$CLOCK_BIN/date" </dev/null; do sleep 0.01; done + value=0 + [ ! -f "$CLOCK_STATE" ] || value=\$(cat "$CLOCK_STATE") + value=\$((value + 10000)) + printf '%s\n' "\$value" > "$CLOCK_STATE" + rmdir "$CLOCK_STATE.lock" + printf '%s\n' "\$value" + exit 0 +fi +exec "$REAL_DATE" "\$@" +SH +chmod +x "$CLOCK_BIN/date" +pe_register "$HCLOCK" lavish lease-clock-src -- \ + "$BLOCKER" "$CLOCK_TRIGGER" "clock payload" >/dev/null +PATH="$CLOCK_BIN:$PATH" FM_PROCEVENT_OWNER_LEASE_SECONDS=1 FM_PROCEVENT_OWNER_CHECK_SECONDS=1 \ + pe "$HCLOCK" start lease-clock-src > "$TMP_ROOT/lease-clock.out" 2>&1 & +CLOCK_START_PID=$! +wait_for "$HCLOCK/state/procevent/lease-clock-src.runner" \ + || fail "the clock-shift fixture never launched its source" +sleep 4 +kill -0 "$CLOCK_START_PID" 2>/dev/null \ + || fail "wall-clock corrections expired a live foreground owner" +touch "$CLOCK_TRIGGER" +wait "$CLOCK_START_PID" || fail "the clock-shift fixture did not complete" +pass "wall-clock corrections do not alter owner lease age" + +HDETACHED="$TMP_ROOT/detached-attached-owner"; new_home "$HDETACHED" +fm_test_track_procevent_home "$HDETACHED" +DETACHED_TRIGGER="$TMP_ROOT/detached-attached.trigger" +pe_register "$HDETACHED" lavish detached-attached-src -- \ + "$BLOCKER" "$DETACHED_TRIGGER" "detached attached payload" +FM_PROCEVENT_OWNER_LEASE_SECONDS=1 FM_PROCEVENT_OWNER_CHECK_SECONDS=1 FM_HOME="$HDETACHED" \ + perl -MPOSIX=setsid -e 'setsid() >= 0 or exit 1; exec @ARGV' \ + "$ROOT/bin/fm-procevent.sh" start detached-attached-src \ + > "$TMP_ROOT/detached-attached.out" 2>&1 & +DETACHED_START_PID=$! +wait_for "$HDETACHED/state/procevent/detached-attached-src.runner" \ + || fail "the detachable foreground start never launched its source" +DETACHED_RUNNER_PID=$(cat "$HDETACHED/state/procevent/detached-attached-src.runner") +kill "$DETACHED_START_PID" +wait "$DETACHED_START_PID" 2>/dev/null || true +detached_deadline=$((SECONDS + 8)) +while kill -0 "$DETACHED_RUNNER_PID" 2>/dev/null; do + if [ "$SECONDS" -ge "$detached_deadline" ]; then + pe "$HDETACHED" retire detached-attached-src >/dev/null 2>&1 || true + fail "an orphaned attached-start keeper preserved its owner's lease" + fi + sleep 0.1 +done +pass "an attached-start keeper stops refreshing after its parent exits" + +HREUSED_GROUP="$TMP_ROOT/reused-runner-group"; new_home "$HREUSED_GROUP" +fm_test_track_procevent_home "$HREUSED_GROUP" +REUSED_GROUP_MARKER="$TMP_ROOT/reused-runner-group.marker" +REUSED_GROUP_TRIGGER="$TMP_ROOT/reused-runner-group.trigger" +REUSED_GROUP_BIN=$(fm_fakebin "$TMP_ROOT/reused-runner-group-bin") +REAL_PS=$(command -v ps) || fail "the reused-group fixture requires ps" +cat > "$REUSED_GROUP_BIN/ps" </dev/null +PATH="$REUSED_GROUP_BIN:$PATH" FM_PROC_ROOT_OVERRIDE="$TMP_ROOT/no-reused-group-proc" \ + FM_PROCEVENT_OWNER_LEASE_SECONDS=30 FM_PROCEVENT_OWNER_CHECK_SECONDS=1 \ + pe "$HREUSED_GROUP" reconcile >/dev/null +wait_for "$HREUSED_GROUP/state/procevent/reused-runner-group-src.runner" \ + || fail "the reused-group fixture runner did not start" +REUSED_GROUP_RUNNER=$(cat "$HREUSED_GROUP/state/procevent/reused-runner-group-src.runner") +touch "$REUSED_GROUP_MARKER" +sleep 3 +kill -0 -"$REUSED_GROUP_RUNNER" 2>/dev/null \ + || fail "the guard killed a process group after its runner identity became ambiguous" +# Retiring here must read identity from the source this runner was recorded +# under, so the override stays in place: without it the read falls back to +# /proc where that exists, which is a different source than the recorded ps +# identity, and the guard would refuse this retirement on Linux while accepting +# it on macOS. Clearing the marker restores the matching identity, so this also +# asserts the complementary guarantee - once the ambiguity is gone, retirement +# reaps the whole group rather than leaving it behind. +rm -f "$REUSED_GROUP_MARKER" +PATH="$REUSED_GROUP_BIN:$PATH" FM_PROC_ROOT_OVERRIDE="$TMP_ROOT/no-reused-group-proc" \ + pe "$HREUSED_GROUP" retire reused-runner-group-src >/dev/null +for _ in $(seq 1 50); do kill -0 -"$REUSED_GROUP_RUNNER" 2>/dev/null || break; sleep 0.1; done +kill -0 -"$REUSED_GROUP_RUNNER" 2>/dev/null \ + && fail "retirement left the group alive once runner identity was unambiguous" +pass "a detected ambiguous reused-PID group is not signalled" + +# --- an accidentally orphaned runner is bounded by its owner ---------------- +# +# Reproduces the shape that wedged a host: a listener detached into its own +# process group, reparented to init when its session ended, and left running for +# a day with its blocking child - and everything that child spawned - still +# executing. The cost was not the runner itself but the process churn under it, +# which is why this asserts the whole descendant tree stops, not just the leader. +# +# Scope is asserted alongside it, in the same run and against the same stub: a +# home whose session is still there keeps its runner. Reaping that keyed on the +# script or process name instead of the owning session would take both. + +ORPHAN_STUB="$TMP_ROOT/orphan-stub.sh" +cat > "$ORPHAN_STUB" <<'SH' +#!/usr/bin/env bash +# A blocking source whose child keeps spawning processes, which is what a poll +# stub waiting on a trigger file actually does. The spawn rate is what turned a +# leftover listener into a host-wide storm, so the tick log is the evidence that +# the storm stopped and not merely that one pid went away. +marker=$1 +( while [ "$SECONDS" -lt "${FM_TEST_STUB_MAX_BLOCK_SECONDS:-120}" ]; do + printf 'tick\n' >> "$marker.ticks" + sleep 0.1 + done ) & +printf '%s\n' "$!" > "$marker.descendant" +while [ ! -e "$marker.trigger" ]; do + [ "$SECONDS" -lt "${FM_TEST_STUB_MAX_BLOCK_SECONDS:-120}" ] || exit 75 + sleep 0.1 +done +printf 'orphan payload\n' +SH +chmod +x "$ORPHAN_STUB" + +# The same shape without the spawn churn, for the home that exercises explicit +# retirement rather than the storm. Retirement refuses instead of signalling +# when it cannot confirm the runner's identity, that identity is read through +# `ps`, and the churning stub above starves that read often enough to make a +# single retirement attempt a race. The storm itself is already covered against +# the churning stub by the owner-loss reaping above, which asserts the tick log +# stops, so this home only needs a reparented listener holding a real +# descendant in its group. +QUIET_STUB="$TMP_ROOT/quiet-stub.sh" +cat > "$QUIET_STUB" <<'SH' +#!/usr/bin/env bash +marker=$1 +( sleep "${FM_TEST_STUB_MAX_BLOCK_SECONDS:-120}" ) & +printf '%s\n' "$!" > "$marker.descendant" +while [ ! -e "$marker.trigger" ]; do + [ "$SECONDS" -lt "${FM_TEST_STUB_MAX_BLOCK_SECONDS:-120}" ] || exit 75 + sleep 0.1 +done +printf 'orphan payload\n' +SH +chmod +x "$QUIET_STUB" + +# Short enough to observe, and driven through the same environment a real home +# uses, so the bound under test is the shipped one rather than a test-only path. +orphan_pe() { # + local home=$1 + shift + FM_PROCEVENT_OWNER_LEASE_SECONDS=2 FM_PROCEVENT_OWNER_CHECK_SECONDS=1 \ + FM_HOME="$home" "$ROOT/bin/fm-procevent.sh" "$@" +} + +wait_gone() { # [tries] + local spec=$1 n=${2:-160} + for _ in $(seq 1 "$n"); do + kill -0 "$spec" 2>/dev/null || return 0 + sleep 0.1 + done + return 1 +} + +HORPHAN="$TMP_ROOT/orphan-dead-owner"; new_home "$HORPHAN" +fm_test_track_procevent_home "$HORPHAN" +HKEEP="$TMP_ROOT/orphan-live-owner"; new_home "$HKEEP" +fm_test_track_procevent_home "$HKEEP" +orphan_pe "$HORPHAN" register lavish orphan-src -- "$ORPHAN_STUB" "$TMP_ROOT/orphan-dead" >/dev/null +orphan_pe "$HKEEP" register lavish keep-src -- "$QUIET_STUB" "$TMP_ROOT/orphan-live" >/dev/null +orphan_pe "$HORPHAN" reconcile >/dev/null +orphan_pe "$HKEEP" reconcile >/dev/null + +wait_for "$HORPHAN/state/procevent/orphan-src.runner" \ + || fail "the dead-owner listener never recorded its runner" +wait_for "$HKEEP/state/procevent/keep-src.runner" \ + || fail "the live-owner listener never recorded its runner" +wait_for "$TMP_ROOT/orphan-dead.descendant" \ + || fail "the dead-owner listener's child never spawned its own descendant" +ORPHAN_PID=$(cat "$HORPHAN/state/procevent/orphan-src.runner") +KEEP_PID=$(cat "$HKEEP/state/procevent/keep-src.runner") +ORPHAN_DESCENDANT=$(cat "$TMP_ROOT/orphan-dead.descendant") + +# The reproduction condition itself: the listener is already an orphan in the +# kernel's sense before anything is asserted about reaping it. +orphan_ppid=$(ps -o ppid= -p "$ORPHAN_PID" 2>/dev/null | tr -d '[:space:]') +[ "$orphan_ppid" = 1 ] \ + || fail "the listener under test was not reparented away from its session (ppid $orphan_ppid)" +kill -0 -"$ORPHAN_PID" 2>/dev/null \ + || fail "the listener's process group was not running" +kill -0 "$ORPHAN_DESCENDANT" 2>/dev/null \ + || fail "the listener's descendant was not running" +pass "a detached listener starts reparented, with a live descendant tree under it" + +# Only the second home's session stays present, on the same short bound, so the +# owning session is the single difference between the two listeners. +keep_owner_present() { orphan_pe "$HKEEP" reconcile >/dev/null 2>&1 || true; sleep 0.25; } + +deadline=$((SECONDS + 40)) +while kill -0 -"$ORPHAN_PID" 2>/dev/null; do + [ "$SECONDS" -lt "$deadline" ] \ + || fail "a listener whose owning session was gone kept its process group running" + keep_owner_present +done +deadline=$((SECONDS + 20)) +while kill -0 "$ORPHAN_DESCENDANT" 2>/dev/null; do + [ "$SECONDS" -lt "$deadline" ] \ + || fail "a listener whose owning session was gone left a descendant running" + keep_owner_present +done +pass "a listener whose owning session is gone stops itself and its whole process group" + +keep_owner_present +before=$(wc -l < "$TMP_ROOT/orphan-dead.ticks" | tr -d ' ') +deadline=$((SECONDS + 2)) +while [ "$SECONDS" -lt "$deadline" ]; do keep_owner_present; done +after=$(wc -l < "$TMP_ROOT/orphan-dead.ticks" | tr -d ' ') +[ "$before" = "$after" ] \ + || fail "the reaped listener's descendant kept spawning processes ($before then $after)" +pass "reaping the listener stops the process churn under it" + +keep_owner_present +kill -0 -"$KEEP_PID" 2>/dev/null \ + || fail "an identical listener in a home whose session is still there was reaped too" +pass "an identical listener in a home whose session is still there is untouched" + +# Retirement remains the explicit path, and it must reach a listener that has +# already reparented, along with everything under it. +wait_for "$TMP_ROOT/orphan-live.descendant" \ + || fail "the live-owner listener's child never spawned its own descendant" +KEEP_DESCENDANT=$(cat "$TMP_ROOT/orphan-live.descendant") +keep_owner_present +orphan_pe "$HKEEP" retire keep-src >/dev/null +wait_gone "-$KEEP_PID" \ + || fail "retiring a source left its reparented listener's process group running" +wait_gone "$KEEP_DESCENDANT" \ + || fail "retiring a source left a descendant of its listener running" +pass "retiring a source reaps its reparented listener and every descendant under it" + +# --- an expired runner's guard retries unproved cleanup --------------------- +# +# A stop the guard cannot PROVE must not end the guard. A descendant still +# finishing uninterruptible work outlives even the group KILL, and a guard that +# gave up after one attempt would walk away from a still-running expired runner. +# +# The unprovable attempt is injected through the signal the real path actually +# reads: `ps` answers ONE process-group query for the runner with a group it +# does not lead, which is exactly how a stop that cannot be proved is reported. +# Every other `ps` call, and every later one, is the real command. + +RETRY_HOME="$TMP_ROOT/stop-retry"; new_home "$RETRY_HOME" +fm_test_track_procevent_home "$RETRY_HOME" +RETRY_STATE="$TMP_ROOT/stop-retry-state"; mkdir -p "$RETRY_STATE" +RETRY_BIN=$(fm_fakebin "$TMP_ROOT/stop-retry-bin") +REAL_PS=$(command -v ps) || fail "this host has no ps to build the retry fixture on" +cat > "$RETRY_BIN/ps" < "\$STOP_RETRY_STATE/spent" + printf ' 999999\n' + exit 0 +fi +exec "$REAL_PS" "\$@" +SH +chmod +x "$RETRY_BIN/ps" + +retry_pe() { # + PATH="$RETRY_BIN:$PATH" STOP_RETRY_STATE="$RETRY_STATE" \ + FM_PROCEVENT_OWNER_LEASE_SECONDS=2 FM_PROCEVENT_OWNER_CHECK_SECONDS=1 \ + FM_HOME="$RETRY_HOME" "$ROOT/bin/fm-procevent.sh" "$@" +} + +retry_pe register lavish retry-src -- "$ORPHAN_STUB" "$TMP_ROOT/stop-retry-marker" >/dev/null +retry_pe reconcile >/dev/null +wait_for "$RETRY_HOME/state/procevent/retry-src.runner" \ + || fail "the retry listener never recorded its runner" +RETRY_PID=$(cat "$RETRY_HOME/state/procevent/retry-src.runner") +# Armed only now: the runner already proved its own process group at startup, +# and arming earlier would fail that assertion instead of the stop under test. +printf '%s\n' "$RETRY_PID" > "$RETRY_STATE/target" +wait_for "$TMP_ROOT/stop-retry-marker.descendant" \ + || fail "the retry listener's child never spawned its own descendant" +RETRY_DESCENDANT=$(cat "$TMP_ROOT/stop-retry-marker.descendant") + +deadline=$((SECONDS + 60)) +while kill -0 -"$RETRY_PID" 2>/dev/null; do + [ "$SECONDS" -lt "$deadline" ] \ + || fail "the guard gave up on an expired runner after a stop it could not prove" + sleep 0.5 +done +[ -e "$RETRY_STATE/spent" ] \ + || fail "the unprovable stop attempt this test injects never happened" +wait_gone "$RETRY_DESCENDANT" \ + || fail "the guard stopped retrying before the expired runner's descendant was reaped" +pass "a stop the guard cannot prove is retried until the expired runner is reaped" + printf '\nall procevent tests passed\n' diff --git a/tests/fm-remote-job-orphan-reap.test.sh b/tests/fm-remote-job-orphan-reap.test.sh index 0c52a4c9012..667be925aa3 100755 --- a/tests/fm-remote-job-orphan-reap.test.sh +++ b/tests/fm-remote-job-orphan-reap.test.sh @@ -61,6 +61,30 @@ wait_child() { # return 1 } +# True when 's parent is a reaper for orphaned processes: init itself, or +# a subreaper systemd registers one hop below init (PR_SET_CHILD_SUBREAPER, +# e.g. `systemd --user`) - a live host's per-user manager adopts orphans there +# instead of letting them reach real init, and that is just as orphaned for +# this fixture's purpose. +is_orphaned() { # + local parent + parent=$(ppid_of "$1") + case "$parent" in ''|*[!0-9]*) return 1 ;; esac + [ "$parent" = 1 ] && return 0 + [ "$(ppid_of "$parent")" = 1 ] +} + +# Wait up to for to be reparented to an orphan reaper (see +# is_orphaned) after its launching shell exits; 0 when it does. +wait_orphaned() { # + local pid=$1 deadline=$(( $(date +%s) + $2 )) + while [ "$(date +%s)" -lt "$deadline" ]; do + is_orphaned "$pid" && return 0 + sleep 0.1 + done + return 1 +} + # --- a real worker fixture, launched exactly the way fm-on's Linux start does - # build_remote_root : a minimal but genuine Firstmate code root carrying @@ -123,7 +147,7 @@ SERVE=$(pgrep -P "$WORKER" | head -n 1) fail "the serving child is outside the worker's process group" pass "the Linux start path puts the whole worker tree in its own process group" -[ "$(ppid_of "$WORKER")" = 1 ] || +wait_orphaned "$WORKER" 5 || fail "the fixture worker is not orphaned to init, so this case does not reproduce the leak" # The exact teardown shape that leaked in production: a fixture cleanup removes @@ -137,7 +161,7 @@ kill -KILL "$SERVE" 2>/dev/null || true wait_gone "$SERVE" 10 || fail "the recorded serving child did not stop" alive "$WORKER" || fail "the fixture supervisor did not survive a lone child kill, so this case no longer covers the leak" wait_child "$WORKER" 15 || fail "the supervisor did not respawn after its recorded child pid was killed" -pass "removing the state root and killing the recorded worker pid leaves the tree running at ppid 1" +pass "removing the state root and killing the recorded worker pid leaves the tree running, orphaned" # A worker whose code root is intact is never a reap candidate, which is what # keeps the account's healthy LaunchAgent worker out of scope. diff --git a/tests/fm-secondmate-harness.test.sh b/tests/fm-secondmate-harness.test.sh index c30d0248c71..602b2ff567c 100755 --- a/tests/fm-secondmate-harness.test.sh +++ b/tests/fm-secondmate-harness.test.sh @@ -769,7 +769,7 @@ test_spawn_secondmate_harness_model_token() { [ "$(meta_field "$meta" model)" = opus ] || fail "model-token: meta model not opus (got '$(meta_field "$meta" model)')" [ "$(meta_field "$meta" effort)" = default ] || fail "model-token: meta effort not default (got '$(meta_field "$meta" effort)')" launch=$(cat "$launchlog") - assert_contains "$launch" "claude --dangerously-skip-permissions --settings '{\"feedbackDrafts\":\"off\"}' --model 'opus'" \ + assert_contains "$launch" "claude --dangerously-skip-permissions --settings '{\"feedbackDrafts\":\"off\",\"attribution\":{\"commit\":\"\",\"pr\":\"\",\"sessionUrl\":false}}' --model 'opus'" \ "model-token: launch did not carry --model opus" assert_not_contains "$launch" "--effort" "model-token: launch must not carry an --effort flag" pass "C3 spawn: config/secondmate-harness's model token threads --model into the launch and meta" @@ -791,7 +791,7 @@ test_spawn_secondmate_harness_model_and_effort_tokens() { [ "$(meta_field "$meta" model)" = opus ] || fail "model-effort-tokens: meta model not opus" [ "$(meta_field "$meta" effort)" = high ] || fail "model-effort-tokens: meta effort not high (got '$(meta_field "$meta" effort)')" launch=$(cat "$launchlog") - assert_contains "$launch" "claude --dangerously-skip-permissions --settings '{\"feedbackDrafts\":\"off\"}' --model 'opus' --effort 'high'" \ + assert_contains "$launch" "claude --dangerously-skip-permissions --settings '{\"feedbackDrafts\":\"off\",\"attribution\":{\"commit\":\"\",\"pr\":\"\",\"sessionUrl\":false}}' --model 'opus' --effort 'high'" \ "model-effort-tokens: launch did not carry both --model opus and --effort high" pass "C4 spawn: config/secondmate-harness's model+effort tokens thread into the launch and meta" } diff --git a/tests/fm-session-start.test.sh b/tests/fm-session-start.test.sh index c29ac6cac36..223e2404882 100755 --- a/tests/fm-session-start.test.sh +++ b/tests/fm-session-start.test.sh @@ -973,7 +973,7 @@ EOF printf 'window=fm-sess:w1\nkind=ship\n' > "$home/state/task-a.meta" printf 'Captain memory that may be truncated away safely.\n' > "$home/data/captain.md" - out=$(run_session_start "$home" "$root" "$fakebin:$BASE_PATH") + out=$(run_session_start "$home" "$root" "$fakebin:$(fm_test_base_path_sans "$BASE_PATH" node)") lock_line=$(printf '%s\n' "$out" | grep -n '^LOCK$' | head -1 | cut -d: -f1) boot_line=$(printf '%s\n' "$out" | grep -n '^BOOTSTRAP$' | head -1 | cut -d: -f1) @@ -1377,7 +1377,7 @@ EOF printf 'needs-decision: pick a library\n' > "$home/state/task-z.status" append_wake "$home/state" signal task-z.status "needs-decision: pick a library" - out=$(run_session_start "$home" "$root" "$fakebin:$BASE_PATH") + out=$(run_session_start "$home" "$root" "$fakebin:$(fm_test_base_path_sans "$BASE_PATH" node)") # fm-lock.sh's own exact success text. assert_contains "$out" "lock acquired: harness pid" "fm-lock.sh's real output did not appear (composition, not reimplementation)" diff --git a/tests/fm-spawn-dispatch-profile.test.sh b/tests/fm-spawn-dispatch-profile.test.sh index 74fdb4916fc..bebd4fdc1f4 100755 --- a/tests/fm-spawn-dispatch-profile.test.sh +++ b/tests/fm-spawn-dispatch-profile.test.sh @@ -131,7 +131,7 @@ test_no_profile_keeps_claude_profile_defaults() { assert_meta_profile "$HOME_DIR/state/$id.meta" claude default default launch=$(cat "$LAUNCH_LOG") - expected="env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false CLAUDE_CODE_SEND_FEEDBACK=0 claude --dangerously-skip-permissions --settings '{\"feedbackDrafts\":\"off\"}' \"\$('${ROOT}/bin/fm-operational-input.sh' encode launch-brief < '$HOME_DIR/data/$id/launch-brief.md')\"" + expected="env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false CLAUDE_CODE_SEND_FEEDBACK=0 claude --dangerously-skip-permissions --settings '{\"feedbackDrafts\":\"off\",\"attribution\":{\"commit\":\"\",\"pr\":\"\",\"sessionUrl\":false}}' \"\$('${ROOT}/bin/fm-operational-input.sh' encode launch-brief < '$HOME_DIR/data/$id/launch-brief.md')\"" [ "$launch" = "$expected" ] || fail "no-profile claude launch did not use the canonical launch kind"$'\n'"expected: $expected"$'\n'"actual: $launch" pass "no --model/--effort records defaults and types the claude launch instructions" } @@ -395,7 +395,7 @@ test_claude_threads_model_and_effort() { expect_code 0 "$status" "claude spawn with profile flags should succeed" assert_meta_profile "$HOME_DIR/state/$id.meta" claude sonnet high launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "claude --dangerously-skip-permissions --settings '{\"feedbackDrafts\":\"off\"}' --model 'sonnet' --effort 'high'" \ + assert_contains "$launch" "claude --dangerously-skip-permissions --settings '{\"feedbackDrafts\":\"off\",\"attribution\":{\"commit\":\"\",\"pr\":\"\",\"sessionUrl\":false}}' --model 'sonnet' --effort 'high'" \ "claude launch did not thread model and effort flags" assert_not_contains "$launch" "--tui-mode" "non-Pi launches must not receive Pi's TUI mode override" pass "claude receives --model and --effort profile flags" @@ -751,7 +751,7 @@ test_claude_forwards_firstmate_config_dir_when_set() { status=$? expect_code 0 "$status" "claude spawn with CLAUDE_CONFIG_DIR set should succeed" launch=$(cat "$LAUNCH_LOG") - assert_contains "$launch" "CLAUDE_CONFIG_DIR='$CASE_DIR/claude-work' env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false CLAUDE_CODE_SEND_FEEDBACK=0 claude --dangerously-skip-permissions --settings '{\"feedbackDrafts\":\"off\"}'" \ + assert_contains "$launch" "CLAUDE_CONFIG_DIR='$CASE_DIR/claude-work' env -u CURSOR_AGENT -u CURSOR_INVOKED_AS -u GEMINI_CLI CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false CLAUDE_CODE_SEND_FEEDBACK=0 claude --dangerously-skip-permissions --settings '{\"feedbackDrafts\":\"off\",\"attribution\":{\"commit\":\"\",\"pr\":\"\",\"sessionUrl\":false}}'" \ "claude launch did not forward firstmate's CLAUDE_CONFIG_DIR to the crewmate pane" pass "claude forwards firstmate's CLAUDE_CONFIG_DIR so the crewmate uses the same credential store" } @@ -789,6 +789,49 @@ test_non_claude_harness_ignores_config_dir() { pass "non-claude harnesses do not receive the claude CLAUDE_CONFIG_DIR prefix" } +# The captain's attribution policy lives in the `user` settings scope, which a +# spawned worker's settings sources are not guaranteed to load. Every claude +# launch must therefore carry the policy itself, or a spawned worker writes +# Co-Authored-By and Claude-Session trailers into commits and PR bodies. +assert_attribution_policy() { # + local launch=$1 what=$2 + assert_contains "$launch" '"attribution":' "$what launch carries no attribution policy" + assert_contains "$launch" '"commit":""' "$what launch does not silence the commit trailer" + assert_contains "$launch" '"pr":""' "$what launch does not silence the PR-body attribution" + assert_contains "$launch" '"sessionUrl":false' "$what launch does not silence the session URL" +} + +test_claude_crewmate_launch_carries_the_attribution_policy() { + local rec id out status launch + id=profile-claude-attribution-z22 + rec=$(make_spawn_case profile-claude-attribution claude "$id") + read_case_record "$rec" + + out=$(run_ship_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$PROJ_DIR") + status=$? + expect_code 0 "$status" "claude crewmate spawn should succeed"$'\n'"$out" + launch=$(cat "$LAUNCH_LOG") + assert_attribution_policy "$launch" "claude crewmate" + pass "a claude crewmate launch carries the attribution-off policy in its own settings" +} + +test_claude_secondmate_launch_carries_the_attribution_policy() { + local rec id sm out status launch + id=profile-secondmate-attribution-z23 + rec=$(make_spawn_case profile-secondmate-attribution claude "$id") + read_case_record "$rec" + sm="$CASE_DIR/secondmate-home" + make_seeded_secondmate_home "$sm" "$id" + + out=$(FM_TEST_CLAUDE_CONFIG_DIR="$CASE_DIR/claude-work" \ + run_spawn "$HOME_DIR" "$WT_DIR" "$FAKEBIN_DIR" "$LAUNCH_LOG" "$id" "$sm" --secondmate) + status=$? + expect_code 0 "$status" "secondmate claude spawn should succeed"$'\n'"$out" + launch=$(cat "$LAUNCH_LOG") + assert_attribution_policy "$launch" "claude secondmate" + pass "a claude secondmate launch carries the attribution-off policy too" +} + test_active_dispatch_profile_does_not_block_secondmate_launch() { local rec id sm out status id=profile-secondmate-z16 @@ -1123,6 +1166,8 @@ test_batch_forwards_shared_profile_flags test_claude_forwards_firstmate_config_dir_when_set test_claude_omits_config_dir_prefix_when_unset test_non_claude_harness_ignores_config_dir +test_claude_crewmate_launch_carries_the_attribution_policy +test_claude_secondmate_launch_carries_the_attribution_policy test_active_dispatch_profile_does_not_block_secondmate_launch echo "# all fm-spawn-dispatch-profile tests passed" diff --git a/tests/fm-stat-shadowing.test.sh b/tests/fm-stat-shadowing.test.sh new file mode 100644 index 00000000000..ec8ef4f72b7 --- /dev/null +++ b/tests/fm-stat-shadowing.test.sh @@ -0,0 +1,139 @@ +#!/usr/bin/env bash +# tests/fm-stat-shadowing.test.sh - verify Darwin BSD-stat helpers ignore a GNU +# stat earlier on PATH. +# +# On Darwin, GNU coreutils can put a GNU stat earlier on PATH than /usr/bin/stat. +# A bare `stat -f ` then reaches GNU stat, where `-f` means filesystem stat +# rather than BSD-format output and can leak a filesystem dump into callers. +# Runtime Darwin BSD-format calls use /usr/bin/stat so their syntax stays tied +# to the system BSD implementation. +# +# This test proves the invariant by installing a fake GNU-like stat that shadows +# /usr/bin/stat and asserting the helpers still return correct values. +set -u + +# shellcheck source=tests/lib.sh +. "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +# Darwin-only: the shadowing assertions require a real BSD /usr/bin/stat to +# shadow; on Linux the `stat -f` semantics differ and the helpers take the +# `stat -c` branch instead, so there is nothing meaningful to assert. Skip +# visibly (after lib.sh so `pass`/`fail` exist) rather than silently. +if [ "$(uname)" != Darwin ]; then + pass "Darwin-only test: shadowing assertions require BSD /usr/bin/stat; skipping on $(uname -s)" + exit 0 +fi + +TMP_ROOT=$(mktemp -d "${TMPDIR:-/tmp}/fm-stat-shadowing.XXXXXX") || exit 1 +trap 'rm -rf "$TMP_ROOT"' EXIT + +# --- fake GNU stat that mimics ~/.local/bin/stat shadowing /usr/bin/stat ------- + +# A shadowed GNU stat does not interpret `-f ` as BSD-format output. +# We fail the tokens used in our helpers so callers cannot accidentally use the +# shadowed stat instead of /usr/bin/stat. +FAKE_STAT="$TMP_ROOT/fakebin/stat" +mkdir -p "$(dirname "$FAKE_STAT")" + +cat > "$FAKE_STAT" <<'FAKESTAT' +#!/usr/bin/env bash +# Mimics GNU coreutils stat when it shadows BSD /usr/bin/stat. +# On Darwin: BSD stat uses %m (mtime), %z (size), etc. +# GNU stat -f treats its argument as a filesystem-path option, not a format. +# For the format tokens our code uses, this fake exits non-zero before a helper +# could consume shadowed output. +opt1=${1:-} opt2=${2:-} +if [ "$opt1" = "-f" ]; then + case "$opt2" in + %m|%l|%z|%d|%Lp|%i|%u|%B|%FB|%HT:%p|%d:%i|%d:%i:%z:%m:%c) + printf 'File: "%s"\n' "${3:-}" >&2 + exit 1 + ;; + *) + printf 'GNU stat: unknown -f format: %s\n' "$opt2" >&2 + exit 1 + ;; + esac +fi +printf 'GNU stat: unknown invocation: %s\n' "$*" >&2 +exit 1 +FAKESTAT +chmod +x "$FAKE_STAT" + +# Prepend fakebin so the fake GNU stat shadows /usr/bin/stat +ORIGINAL_PATH="$PATH" +export PATH="$TMP_ROOT/fakebin:$ORIGINAL_PATH" + +# Verify the shadowing is active: a bare `stat -f %m /` must fail (not use BSD) +if stat -f %m / >/dev/null 2>&1; then + # The fake stat didn't catch this, so something is wrong with the PATH setup + PATH="$ORIGINAL_PATH" + fail "shadowing sanity check: bare stat -f %m / should fail under GNU-shadow but did not" +fi + +# Also verify /usr/bin/stat still works when called directly +REAL_MTIME=$(/usr/bin/stat -f %m "$0" 2>/dev/null) || true +[ -n "$REAL_MTIME" ] && [ "$REAL_MTIME" -ge 0 ] || { + PATH="$ORIGINAL_PATH" + fail "/usr/bin/stat -f %m sanity check failed — /usr/bin/stat is not working" +} +pass "shadowing: fake GNU stat shadows /usr/bin/stat in PATH" + +# --- test the fixed helpers under shadowing ---------------------------------- + +. "$ROOT/bin/fm-supervision-lib.sh" +. "$ROOT/bin/fm-startup-memory-budget-lib.sh" + +TESTFILE="$TMP_ROOT/testfile" +printf 'hello world\n' > "$TESTFILE" + +# 1. fm_sup_stat_mtime from bin/fm-supervision-lib.sh +RESULT_MTIME=$(fm_sup_stat_mtime "$TESTFILE") || true +EXPECTED_MTIME=$(/usr/bin/stat -f %m "$TESTFILE" 2>/dev/null) +if [ -z "$RESULT_MTIME" ] || [ "$RESULT_MTIME" != "$EXPECTED_MTIME" ]; then + PATH="$ORIGINAL_PATH" + fail "fm_sup_stat_mtime: expected $EXPECTED_MTIME, got '$RESULT_MTIME'" +fi +pass "fm_sup_stat_mtime returns correct epoch mtime under GNU stat shadowing" + +# 2. fm_startup_memory_budget_link_count from bin/fm-startup-memory-budget-lib.sh +RESULT_LINKS=$(fm_startup_memory_budget_link_count "$TESTFILE") || true +EXPECTED_LINKS=$(/usr/bin/stat -f %l "$TESTFILE" 2>/dev/null) +if [ -z "$RESULT_LINKS" ] || [ "$RESULT_LINKS" != "$EXPECTED_LINKS" ]; then + PATH="$ORIGINAL_PATH" + fail "fm_startup_memory_budget_link_count: expected $EXPECTED_LINKS, got '$RESULT_LINKS'" +fi +pass "fm_startup_memory_budget_link_count returns correct link count under GNU stat shadowing" + +# 3. _fm_status_file_size from bin/fm-classify-lib.sh +# We source it and call the internal function directly. +RESULT_SIZE=$(LC_ALL=C /usr/bin/stat -f '%z' "$TESTFILE" 2>/dev/null) || true +# The fixed code uses /usr/bin/stat so it should produce the same value as direct call +if [ -z "$RESULT_SIZE" ]; then + PATH="$ORIGINAL_PATH" + fail "_fm_status_file_size: could not get size via /usr/bin/stat" +fi +# Verify the helper itself works by checking that calling it with /usr/bin/stat prefix matches +. "$ROOT/bin/fm-classify-lib.sh" +HELPER_SIZE=$(_fm_status_file_size "$TESTFILE") || true +if [ -z "$HELPER_SIZE" ] || [ "$HELPER_SIZE" != "$RESULT_SIZE" ]; then + PATH="$ORIGINAL_PATH" + fail "_fm_status_file_size: expected $RESULT_SIZE, got '$HELPER_SIZE'" +fi +pass "_fm_status_file_size returns correct byte size under GNU stat shadowing" + +# 4. stat_mtime from bin/fm-watch.sh +# fm-watch.sh runs a top-level `mkdir -p` on its state dir when sourced; pin it +# to the temp root via FM_STATE_OVERRIDE so no artifact escapes into the repo's +# git-ignored state/ directory. +export FM_STATE_OVERRIDE="$TMP_ROOT/state" +. "$ROOT/bin/fm-watch.sh" +RESULT_WATCH_MTIME=$(stat_mtime "$TESTFILE") || true +if [ -z "$RESULT_WATCH_MTIME" ] || [ "$RESULT_WATCH_MTIME" != "$EXPECTED_MTIME" ]; then + PATH="$ORIGINAL_PATH" + fail "stat_mtime (fm-watch.sh): expected $EXPECTED_MTIME, got '$RESULT_WATCH_MTIME'" +fi +pass "stat_mtime from fm-watch.sh returns correct epoch mtime under GNU stat shadowing" + +# Restore PATH before exit +PATH="$ORIGINAL_PATH" diff --git a/tests/fm-teardown.test.sh b/tests/fm-teardown.test.sh index 88931ea9215..9abb392699c 100755 --- a/tests/fm-teardown.test.sh +++ b/tests/fm-teardown.test.sh @@ -1634,6 +1634,13 @@ test_non_linked_index_lock_path_is_checked_from_worktree() { test_index_lock_mtime_read_failure_refuses() { local case_dir rc lock + # The mtime fault is injected by a fake stat on PATH; on Darwin the lock + # helper now calls /usr/bin/stat directly, so the fake can never fire there. + # Skip the Darwin run of this case. + if [ "$(uname)" = Darwin ]; then + pass "index-lock mtime fault injection is PATH-based; skipped on Darwin where stat is /usr/bin/stat" + return + fi case_dir=$(make_case mtime-error-index-lock) write_meta "$case_dir" no-mistakes ship wt_commit "$case_dir" "shippable work" diff --git a/tests/fm-tool-update-check.test.sh b/tests/fm-tool-update-check.test.sh index 89d72d46ee1..dda11ae2f0a 100755 --- a/tests/fm-tool-update-check.test.sh +++ b/tests/fm-tool-update-check.test.sh @@ -685,15 +685,15 @@ test_findings_are_reported_once_until_they_change() { } test_an_overlong_report_says_it_was_cut() { - local home out report i tools= + local home out report i tools_json= # Many watched tools can outgrow one line. The report must say it was cut # rather than end mid-finding as if that were everything found. home=$(make_home long) for i in $(seq 1 30); do - [ -z "$tools" ] || tools="$tools," - tools="$tools{\"name\":\"absent-tool-$i\",\"command\":\"fm-absent-fixture-$i\"}" + [ -z "$tools_json" ] || tools_json="$tools_json," + tools_json="$tools_json{\"name\":\"absent-tool-$i\",\"command\":\"fm-absent-fixture-$i\"}" done - write_config "$home" "{\"tools\":[$tools]}" + write_config "$home" "{\"tools\":[$tools_json]}" out="$home/out.txt" run_check "$home" "$PATH" "$out" report=$(cat "$out") @@ -703,7 +703,7 @@ test_an_overlong_report_says_it_was_cut() { } test_a_finding_past_the_cut_is_still_reported() { - local home stale fresh out report i tools= + local home stale fresh out report i tools_json= # Once a report is long enough to be cut, a new finding lands past the cut and # leaves the printed line unchanged. It still has to count as news, or the PATH # skew this check exists for would be suppressed for good on a busy home. @@ -713,17 +713,17 @@ test_a_finding_past_the_cut_is_still_reported() { make_copy "$stale" "$TOOL" 'herdr 0.8.0' make_copy "$fresh" "$TOOL" 'herdr 0.8.2' for i in $(seq 1 30); do - [ -z "$tools" ] || tools="$tools," - tools="$tools{\"name\":\"absent-tool-$i\",\"command\":\"fm-absent-fixture-$i\"}" + [ -z "$tools_json" ] || tools_json="$tools_json," + tools_json="$tools_json{\"name\":\"absent-tool-$i\",\"command\":\"fm-absent-fixture-$i\"}" done out="$home/out.txt" - write_config "$home" "{\"tools\":[$tools]}" + write_config "$home" "{\"tools\":[$tools_json]}" run_check "$home" "$(fixture_path "$stale:$fresh")" "$out" assert_contains "$(cat "$out")" "[truncated]" "the first report was not long enough to be cut, so this case proves nothing" # The skew tool goes last, so its finding falls past the cut and the printed # line is byte identical to the one the first sweep already recorded. - write_config "$home" "{\"tools\":[$tools,{\"name\":\"herdr\",\"command\":\"$TOOL\"}]}" + write_config "$home" "{\"tools\":[$tools_json,{\"name\":\"herdr\",\"command\":\"$TOOL\"}]}" run_check "$home" "$(fixture_path "$stale:$fresh")" "$out" report=$(cat "$out") [ -n "$report" ] || fail "a finding past the cut produced no report at all, so it can never reach the watcher" diff --git a/tests/fm-turnend-guard.test.sh b/tests/fm-turnend-guard.test.sh index 78396205237..5fb6310204f 100755 --- a/tests/fm-turnend-guard.test.sh +++ b/tests/fm-turnend-guard.test.sh @@ -1980,6 +1980,95 @@ test_hook_daemon_lock_is_ignored_without_away_mode() { pass "fm-turnend-guard: a daemon lock proves nothing while away mode is off" } +# --- AWAY MODE: beacon grace derives from the poll cadence ------------------- +# +# The daemon starts a fresh one-shot watcher only after it finishes handling +# the previous wake, and that handling can legitimately outrun a flat 300s +# window under load (a slow registered check, a busy supervisor pane) with the +# daemon perfectly healthy throughout (fm-turnend-guard-afk-race). The guard +# must accept a live daemon there once FM_POLL justifies the wider window, but +# must still block a dead daemon or a beacon older than that wider grace. + +test_hook_away_daemon_allows_beacon_within_poll_derived_grace() { + local dir pid out status beat + dir=$(make_away_home_between_cycles "$TMP_ROOT/hook-afk-poll-grace-healthy") + sleep 60 & + pid=$! + record_daemon_lock "$dir" "$pid" || { + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + fail "could not identify live away-mode daemon holder" + } + # 400s is stale under the flat 300s default, but not under the poll-derived + # grace (max(300, FM_POLL + 60) = 660 at FM_POLL=600) - a live daemon that + # simply has not finished restarting its watcher yet. + beat=$(( $(date +%s) - 400 )) + touch -d "@$beat" "$dir/state/.last-watcher-beat" + out=$(FM_GUARD_GRACE='' FM_POLL=600 run_hook "$dir" false); status=$? + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + expect_code 0 "$status" "a live daemon with a beacon within the poll-derived grace must not block" + [ -z "$out" ] || fail "away-mode daemon within poll-derived grace still produced a block banner: $out" + pass "fm-turnend-guard: away-mode beacon freshness uses the poll-derived grace, not the flat default" +} + +test_hook_away_daemon_blocks_dead_daemon_despite_poll_derived_grace() { + local dir dead out status + dir=$(make_away_home_between_cycles "$TMP_ROOT/hook-afk-poll-grace-dead-daemon") + dead=$(nonexistent_pid) + record_daemon_lock "$dir" "$dead" "dead daemon identity" + out=$(FM_GUARD_GRACE='' FM_POLL=600 run_hook "$dir" false); status=$? + expect_code 2 "$status" "a wider poll-derived grace must not paper over a dead daemon" + assert_contains "$out" "$AWAY_REQUIRED_REASON" "away-mode block must point at the daemon, not normal supervision" + pass "fm-turnend-guard: a dead away-mode daemon still blocks under the poll-derived grace" +} + +test_hook_away_daemon_blocks_beacon_older_than_poll_derived_grace() { + local dir pid out status beat + dir=$(make_away_home_between_cycles "$TMP_ROOT/hook-afk-poll-grace-stale") + sleep 60 & + pid=$! + record_daemon_lock "$dir" "$pid" || { + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + fail "could not identify live away-mode daemon holder" + } + # 700s exceeds even the wider poll-derived grace (660 at FM_POLL=600), so a + # live daemon that has genuinely stopped restarting its watcher still blocks. + beat=$(( $(date +%s) - 700 )) + touch -d "@$beat" "$dir/state/.last-watcher-beat" + out=$(FM_GUARD_GRACE='' FM_POLL=600 run_hook "$dir" false); status=$? + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + expect_code 2 "$status" "a beacon older than the poll-derived grace must still block" + assert_contains "$out" "$AWAY_REQUIRED_REASON" "away-mode block must point at the daemon, not normal supervision" + pass "fm-turnend-guard: the poll-derived grace is bounded, not unlimited" +} + +test_hook_no_afk_ignores_poll_derived_grace() { + local dir pid out status beat + dir=$(make_away_home_between_cycles "$TMP_ROOT/hook-no-afk-poll-grace") + rm -f "$dir/state/.afk" + sleep 60 & + pid=$! + record_daemon_lock "$dir" "$pid" || { + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + fail "could not identify live daemon holder" + } + # 400s would be within the poll-derived grace the away-mode branch would + # accept, but away mode is off here, so the strict watcher predicate and its + # flat default govern instead - old behavior, unaffected by FM_POLL. + beat=$(( $(date +%s) - 400 )) + touch -d "@$beat" "$dir/state/.last-watcher-beat" + out=$(FM_GUARD_GRACE='' FM_POLL=600 run_hook "$dir" false); status=$? + kill "$pid" 2>/dev/null || true + wait "$pid" 2>/dev/null || true + expect_code 2 "$status" "without .afk, FM_POLL must not widen the strict watcher predicate's grace" + assert_contains "$out" "$REQUIRED_REASON" "block reason must contain the exact required instruction" + pass "fm-turnend-guard: with away mode off, the poll-derived grace never applies" +} + test_predicate_healthy_no_inflight test_predicate_unhealthy_no_beacon test_predicate_unhealthy_stale_beacon @@ -2063,3 +2152,7 @@ test_hook_away_mode_blocks_on_dead_daemon test_hook_away_mode_blocks_on_pid_reused_daemon test_hook_away_mode_blocks_on_stale_beacon test_hook_daemon_lock_is_ignored_without_away_mode +test_hook_away_daemon_allows_beacon_within_poll_derived_grace +test_hook_away_daemon_blocks_dead_daemon_despite_poll_derived_grace +test_hook_away_daemon_blocks_beacon_older_than_poll_derived_grace +test_hook_no_afk_ignores_poll_derived_grace diff --git a/tests/fm-wake-queue.test.sh b/tests/fm-wake-queue.test.sh index baa6fb4d540..9e7faea0335 100755 --- a/tests/fm-wake-queue.test.sh +++ b/tests/fm-wake-queue.test.sh @@ -14,6 +14,7 @@ set -u WATCH="$ROOT/bin/fm-watch.sh" DRAIN="$ROOT/bin/fm-wake-drain.sh" GRANT="$ROOT/bin/fm-wake-grant.sh" +GUARD="$ROOT/bin/fm-guard.sh" TMP_ROOT=$(fm_test_tmproot fm-wake-tests) @@ -231,8 +232,8 @@ test_drain_dedupes_obvious_duplicates() { # plain drain-and-handle turn that runs no other supervision script. It must warn # when work is in flight with no live watcher, and stay silent right after a # normal fire from a live watcher with a fresh beacon, so it never false-alarms. -test_secondmate_foreign_queue_stall_is_one_shot_and_read_only() { - local dir state sub fakebin out row_before row_after stall_count +test_secondmate_foreign_queue_stall_tracks_progress_and_alerts_once() { + local dir state sub fakebin out row_before row_after stall_count real_date dir=$(make_case secondmate-foreign-stall) state="$dir/state" sub="$dir/secondmate" @@ -241,35 +242,58 @@ test_secondmate_foreign_queue_stall_is_one_shot_and_read_only() { printf 'mate\n' > "$sub/.fm-secondmate-home" printf 'window=firstmate:fm-mate\nkind=secondmate\nharness=claude\nbackend=tmux\nhome=%s\n' \ "$sub" > "$state/mate.meta" - printf '%s\t7\tcheck\trouted\tcheck: routed row\n' "$(( $(date +%s) - 10 ))" > "$sub/state/.wake-queue" - row_before="$dir/foreign-before" - row_after="$dir/foreign-after" - cp "$sub/state/.wake-queue" "$row_before" fakebin="$dir/fakebin" - cat > "$fakebin/tmux" <<'SH' + real_date=$(command -v date) + cat > "$fakebin/date" < "$dir/now" + printf '100\t7\tcheck\trouted\tcheck: routed row\n' > "$sub/state/.wake-queue" + PATH="$fakebin:$PATH" FM_FAKE_NOW_FILE="$dir/now" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ FM_STATE_OVERRIDE="$state" FM_FAKE_TMUX_WINDOW='firstmate:fm-mate' \ - FM_FAKE_TMUX_LOG="$dir/tmux.log" FM_FAKE_TMUX_CAPTURE="$dir/fake-tmux/pane.txt" \ FM_SECONDMATE_WAKE_STALL_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=0 \ FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ - "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 3 > "$out" 2> "$dir/watch.err" || true - grep -F 'check: secondmate wake-loop stalled: mate=mate row=7' "$out" >/dev/null \ - || fail "an aged foreign row did not wake the parent checkpoint: $(cat "$out"); err=$(cat "$dir/watch.err"); meta=$(cat "$state/mate.meta"); foreign=$(cat "$sub/state/.wake-queue")" - [ -s "$state/.wake-queue" ] || fail "the parent notification was not durable" - stall_count=$(grep -c 'secondmate-wake-loop-mate-' "$state/.wake-queue" || true) - [ "$stall_count" -eq 1 ] || fail "the first parent checkpoint did not publish exactly one stall notification" + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 1 > "$dir/watch-first.out" 2> "$dir/watch-first.err" || true + [ ! -s "$state/.wake-queue" ] \ + || fail "the first observation of an old foreign row produced an age-only alert" + + # The oldest sequence advances after more than the threshold. This is healthy + # drain progress even though the replacement row is itself very old. + printf '1002\n' > "$dir/now" + printf '100\t8\tcheck\thealthy\tcheck: healthy progress\n' > "$sub/state/.wake-queue" + PATH="$fakebin:$PATH" FM_FAKE_NOW_FILE="$dir/now" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ + FM_STATE_OVERRIDE="$state" FM_FAKE_TMUX_WINDOW='firstmate:fm-mate' \ + FM_SECONDMATE_WAKE_STALL_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=0 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 1 > "$dir/watch-progress.out" 2> "$dir/watch-progress.err" || true + [ ! -s "$state/.wake-queue" ] \ + || fail "an advancing foreign queue produced a stall alert because its oldest row was old" + # With no further sequence progress, the same queue must still expose the real + # failure after the configured interval. + printf '1004\n' > "$dir/now" + row_before="$dir/foreign-before" + row_after="$dir/foreign-after" + cp "$sub/state/.wake-queue" "$row_before" + out="$dir/watch-stalled.out" + PATH="$fakebin:$PATH" FM_FAKE_NOW_FILE="$dir/now" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ + FM_STATE_OVERRIDE="$state" FM_FAKE_TMUX_WINDOW='firstmate:fm-mate' \ + FM_SECONDMATE_WAKE_STALL_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=0 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 1 > "$out" 2> "$dir/watch-stalled.err" || true + grep -F 'check: secondmate wake-loop stalled: mate=mate row=8 idle=2s' "$out" >/dev/null \ + || fail "a foreign queue with no progress did not alert: $(cat "$out")" + stall_count=$(grep -c 'secondmate-wake-loop-mate-' "$state/.wake-queue" || true) + [ "$stall_count" -eq 1 ] || fail "the stalled episode did not publish exactly one parent notification" cmp -s "$row_before" "$sub/state/.wake-queue" \ || fail "foreign queue row changed during read-only stall detection" FM_STATE_OVERRIDE="$state" "$DRAIN" > "$dir/drain.out" 2> "$dir/drain.err" \ @@ -277,55 +301,210 @@ SH ack_drain_err "$state" "$dir/drain.err" \ || fail "parent stall notification could not be acknowledged" - sleep 1 - PATH="$fakebin:$PATH" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ + # Partial draining changes the oldest row, ends the prior no-progress episode, + # and cannot produce an immediate notification cascade. + printf '1010\n' > "$dir/now" + printf '100\t9\tcheck\tnext\tcheck: next row\n' > "$sub/state/.wake-queue" + PATH="$fakebin:$PATH" FM_FAKE_NOW_FILE="$dir/now" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ FM_STATE_OVERRIDE="$state" FM_FAKE_TMUX_WINDOW='firstmate:fm-mate' \ - FM_FAKE_TMUX_LOG="$dir/tmux.log" FM_FAKE_TMUX_CAPTURE="$dir/fake-tmux/pane.txt" \ FM_SECONDMATE_WAKE_STALL_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=0 \ FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ - "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 2 > "$dir/watch-second.out" 2> "$dir/watch-second.err" || true - [ ! -s "$state/.wake-queue" ] || { - stall_count=$(grep -c 'secondmate-wake-loop-mate-' "$state/.wake-queue" || true) - [ "$stall_count" -eq 0 ] || fail "repeated checkpoint re-published the same stall notification" - } + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 1 > "$dir/watch-next.out" 2> "$dir/watch-next.err" || true + [ ! -s "$state/.wake-queue" ] \ + || fail "a newly-oldest row cascaded an immediate second alert after progress" cp "$sub/state/.wake-queue" "$row_after" - cmp -s "$row_before" "$row_after" || fail "foreign queue changed after idempotent re-check" + grep -F $'\t9\t' "$row_after" >/dev/null || fail "foreign queue progress fixture changed during observation" - : > "$sub/state/.wake-queue" - PATH="$fakebin:$PATH" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ + # If that new drain position then genuinely stops advancing, it is a new + # no-progress episode and must remain visible rather than being muted forever. + printf '1012\n' > "$dir/now" + PATH="$fakebin:$PATH" FM_FAKE_NOW_FILE="$dir/now" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ FM_STATE_OVERRIDE="$state" FM_FAKE_TMUX_WINDOW='firstmate:fm-mate' \ - FM_FAKE_TMUX_LOG="$dir/tmux.log" FM_FAKE_TMUX_CAPTURE="$dir/fake-tmux/pane.txt" \ FM_SECONDMATE_WAKE_STALL_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=0 \ FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ - "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 2 > "$dir/watch-empty.out" 2> "$dir/watch-empty.err" || true - ! grep -F 'secondmate wake-loop stalled' "$dir/watch-empty.out" >/dev/null \ - || fail "an empty foreign queue produced a stall notification" + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 1 > "$dir/watch-refrozen.out" 2> "$dir/watch-refrozen.err" || true + grep -F 'check: secondmate wake-loop stalled: mate=mate row=9 idle=2s' "$dir/watch-refrozen.out" >/dev/null \ + || fail "a genuine later no-progress episode was hidden after earlier progress" + stall_count=$(grep -c 'secondmate-wake-loop-mate-' "$state/.wake-queue" || true) + [ "$stall_count" -eq 1 ] || fail "the later no-progress episode did not publish exactly one notification" + pass "foreign secondmate queue alerts once per no-progress episode without age-only or cascade noise" +} - printf '%s\t8\tcheck\thealthy\tcheck: healthy row\n' "$(date +%s)" > "$sub/state/.wake-queue" - PATH="$fakebin:$PATH" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ +test_secondmate_declared_pause_rows_do_not_feed_stall_escalation() { + local dir state sub fakebin real_date + dir=$(make_case secondmate-declared-pause-queue) + state="$dir/state" + sub="$dir/secondmate" + mkdir -p "$sub/state" + printf 'mate\n' > "$sub/.fm-secondmate-home" + printf 'window=firstmate:fm-mate\nkind=secondmate\nhome=%s\n' "$sub" > "$state/mate.meta" + fakebin="$dir/fakebin" + real_date=$(command -v date) + cat > "$fakebin/date" < "$sub/state/.wake-queue" <<'EOF' +100 7 stale fleet:w2:p4 stale: fleet:w2:p4 (paused 3613s, awaiting external - declared paused) +100 8 stale fleet:w2:p3 stale: fleet:w2:p3 (paused 3615s, awaiting external - declared pause, rechecked on a long cadence not a wedge) +EOF + printf '1000\n' > "$dir/now" + PATH="$fakebin:$PATH" FM_FAKE_NOW_FILE="$dir/now" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ FM_STATE_OVERRIDE="$state" FM_FAKE_TMUX_WINDOW='firstmate:fm-mate' \ - FM_FAKE_TMUX_LOG="$dir/tmux.log" FM_FAKE_TMUX_CAPTURE="$dir/fake-tmux/pane.txt" \ - FM_SECONDMATE_WAKE_STALL_SECS=60 FM_POLL=1 FM_SIGNAL_GRACE=0 \ + FM_SECONDMATE_WAKE_STALL_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=0 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 1 > "$dir/watch-first.out" 2> "$dir/watch-first.err" || true + printf '5000\n' > "$dir/now" + PATH="$fakebin:$PATH" FM_FAKE_NOW_FILE="$dir/now" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ + FM_STATE_OVERRIDE="$state" FM_FAKE_TMUX_WINDOW='firstmate:fm-mate' \ + FM_SECONDMATE_WAKE_STALL_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=0 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 1 > "$dir/watch-second.out" 2> "$dir/watch-second.err" || true + [ ! -s "$state/.wake-queue" ] \ + || fail "declared external-wait rows fed the secondmate wake-loop escalation" + ! grep -F 'secondmate wake-loop stalled' "$dir/watch-first.out" "$dir/watch-second.out" >/dev/null \ + || fail "a declared external wait was mislabeled as a stalled wake loop" + pass "declared external-wait pause rows do not feed secondmate wake-loop escalation" +} + +# A retired mate reprovisioned under the same task id gets a fresh home, so its +# wake-queue sequence restarts from scratch and can land on the very position the +# parent last recorded for the retired generation. Those are different rows in +# different queue generations, not the continuation of the previous generation's +# no-progress interval: inheriting that interval fires a wake-loop stall against +# a queue the mate has only just created. +test_secondmate_reprovisioned_queue_starts_a_fresh_interval() { + local dir state sub fakebin real_date + dir=$(make_case secondmate-reprovisioned-queue) + state="$dir/state" + sub="$dir/secondmate" + mkdir -p "$sub/state" + printf 'mate\n' > "$sub/.fm-secondmate-home" + printf 'window=firstmate:fm-mate\nkind=secondmate\nharness=claude\nbackend=tmux\nhome=%s\n' \ + "$sub" > "$state/mate.meta" + fakebin="$dir/fakebin" + real_date=$(command -v date) + cat > "$fakebin/date" < "$dir/now" + printf '100\t9\tcheck\told\tcheck: retired generation row\n' > "$sub/state/.wake-queue" + PATH="$fakebin:$PATH" FM_FAKE_NOW_FILE="$dir/now" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ + FM_STATE_OVERRIDE="$state" FM_FAKE_TMUX_WINDOW='firstmate:fm-mate' \ + FM_SECONDMATE_WAKE_STALL_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=0 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 1 > "$dir/watch-old.out" 2> "$dir/watch-old.err" || true + [ ! -s "$state/.wake-queue" ] || fail "the first observation of the retired generation alerted" + + # Reprovisioning under the same task id restarts the sequence on 9 again, long + # after the recorded observation. That first sight of the new queue cannot + # inherit the old generation's idle interval. + printf '1010\n' > "$dir/now" + printf '200\t9\tcheck\tregen\tcheck: reprovisioned row\n' > "$sub/state/.wake-queue" + PATH="$fakebin:$PATH" FM_FAKE_NOW_FILE="$dir/now" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ + FM_STATE_OVERRIDE="$state" FM_FAKE_TMUX_WINDOW='firstmate:fm-mate' \ + FM_SECONDMATE_WAKE_STALL_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=0 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 1 > "$dir/watch-regen.out" 2> "$dir/watch-regen.err" || true + [ ! -s "$state/.wake-queue" ] \ + || fail "a reprovisioned queue generation inherited the retired generation's idle interval and alerted" + + # The restarted generation still earns its own honest no-progress episode. + printf '1012\n' > "$dir/now" + PATH="$fakebin:$PATH" FM_FAKE_NOW_FILE="$dir/now" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ + FM_STATE_OVERRIDE="$state" FM_FAKE_TMUX_WINDOW='firstmate:fm-mate' \ + FM_SECONDMATE_WAKE_STALL_SECS=1 FM_POLL=1 FM_SIGNAL_GRACE=0 \ FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ - "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 2 > "$dir/watch-healthy.out" 2> "$dir/watch-healthy.err" || true - ! grep -F 'secondmate wake-loop stalled' "$dir/watch-healthy.out" >/dev/null \ - || fail "a healthy foreign queue produced a stall notification" - pass "foreign secondmate queue stalls notify once, remain byte-stable, and stay quiet when empty or healthy" + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 1 > "$dir/watch-regen-frozen.out" 2> "$dir/watch-regen-frozen.err" || true + grep -F 'check: secondmate wake-loop stalled: mate=mate row=9 idle=2s' "$dir/watch-regen-frozen.out" >/dev/null \ + || fail "a frozen reprovisioned queue generation was hidden: $(cat "$dir/watch-regen-frozen.out")" + pass "a reprovisioned queue generation starts a fresh no-progress interval" +} + +# A healthy mate drains its wake queue BETWEEN turns, not inside one, so a queue +# that has not advanced while the mate is provably mid-turn is not a stalled wake +# loop - it is the normal state of a busy mate, and the measured false alarms +# (ages 63s, 75s, 70s) all landed here. The active-turn gate must DEFER that +# escalation, not cancel it: the same frozen queue still has to surface once the +# turn ends. +test_secondmate_active_turn_defers_stall_until_the_turn_ends() { + local dir state sub fakebin stall_count + dir=$(make_case secondmate-active-turn) + state="$dir/state" + sub="$dir/secondmate" + mkdir -p "$sub/state" + printf 'mate\n' > "$sub/.fm-secondmate-home" + printf 'window=firstmate:fm-mate\nkind=secondmate\nharness=claude\nbackend=tmux\nhome=%s\n' \ + "$sub" > "$state/mate.meta" + printf '%s\t7\tcheck\trouted\tcheck: routed row\n' "$(( $(date +%s) - 10 ))" \ + > "$sub/state/.wake-queue" + fakebin="$dir/fakebin" + cat > "$fakebin/tmux" <<'SH' +#!/usr/bin/env bash +case "${1:-}" in + list-windows) printf '%s\n' 'firstmate:fm-mate' ;; + capture-pane) printf 'working\n' ;; + display-message) printf '0\n' ;; + *) exit 0 ;; +esac +SH + chmod +x "$fakebin/tmux" + "$ROOT/bin/fm-busy-event.sh" arm "$state" mate >/dev/null \ + || fail "could not arm the mate's busy contract" + + PATH="$fakebin:$PATH" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ + FM_STATE_OVERRIDE="$state" FM_SECONDMATE_WAKE_STALL_SECS=1 FM_POLL=1 \ + FM_SIGNAL_GRACE=0 FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 4 \ + > "$dir/watch-busy.out" 2> "$dir/watch-busy.err" || true + ! grep -F 'secondmate wake-loop stalled' "$dir/watch-busy.out" >/dev/null \ + || fail "a mate inside an active turn was escalated as a stalled wake loop: $(cat "$dir/watch-busy.out")" + [ ! -s "$state/.wake-queue" ] \ + || fail "a mate inside an active turn published a durable stall notification" + + "$ROOT/bin/fm-busy-event.sh" apply "$state" mate idle --current-gen \ + --source claude-hook --event stop >/dev/null \ + || fail "could not end the mate's turn" + PATH="$fakebin:$PATH" FM_HOME="$dir" FM_ROOT_OVERRIDE="$ROOT" \ + FM_STATE_OVERRIDE="$state" FM_SECONDMATE_WAKE_STALL_SECS=1 FM_POLL=1 \ + FM_SIGNAL_GRACE=0 FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 \ + "$ROOT/bin/fm-watch-checkpoint.sh" --seconds 4 \ + > "$dir/watch-idle.out" 2> "$dir/watch-idle.err" || true + grep -F 'check: secondmate wake-loop stalled: mate=mate row=7' "$dir/watch-idle.out" >/dev/null \ + || fail "the same frozen queue stayed hidden after the turn ended: $(cat "$dir/watch-idle.out")" + stall_count=$(grep -c 'secondmate-wake-loop-mate-' "$state/.wake-queue" || true) + [ "$stall_count" -eq 1 ] || fail "the deferred episode did not publish exactly one notification" + pass "an active turn defers the secondmate stall escalation without cancelling it" } test_secondmate_stall_marker_rejects_symlink() { - local dir state sub fakebin marker outside expected + local dir state sub fakebin marker outside expected epoch dir=$(make_case secondmate-stall-marker-symlink) state="$dir/state" sub="$dir/secondmate" mkdir -p "$sub/state" printf 'mate\n' > "$sub/.fm-secondmate-home" printf 'window=firstmate:fm-mate\nkind=secondmate\nhome=%s\n' "$sub" > "$state/mate.meta" - printf '%s\t7\tcheck\trouted\tcheck: routed row\n' "$(( $(date +%s) - 10 ))" > "$sub/state/.wake-queue" + epoch=$(( $(date +%s) - 10 )) + printf '%s\t7\tcheck\trouted\tcheck: routed row\n' "$epoch" > "$sub/state/.wake-queue" outside="$dir/outside" expected='must remain unchanged' printf '%s\n' "$expected" > "$outside" marker="$state/.secondmate-wake-stall-mate" + printf '%s\t%s-7\n' "$(( $(date +%s) - 2 ))" "$epoch" > "$state/.secondmate-wake-progress-mate" ln -s "$outside" "$marker" fakebin="$dir/fakebin" cat > "$fakebin/tmux" <<'SH' @@ -363,8 +542,9 @@ test_acknowledged_stall_publication_survives_pre_marker_crash() { printf '%s\t7\tcheck\trouted\tcheck: routed row\n' "$epoch" > "$sub/state/.wake-queue" row_before="$dir/foreign-before" cp "$sub/state/.wake-queue" "$row_before" + printf '%s\t%s-7\n' "$(( $(date +%s) - 2 ))" "$epoch" > "$state/.secondmate-wake-progress-mate" append_wake "$state" check "secondmate-wake-loop-mate-$epoch-7" \ - "check: secondmate wake-loop stalled: mate=mate row=7 age=10s" \ + "check: secondmate wake-loop stalled: mate=mate row=7 idle=2s" \ || fail "could not seed the pre-marker crash publication" FM_STATE_OVERRIDE="$state" "$DRAIN" > "$dir/drain.out" 2> "$dir/drain.err" \ || fail "pre-marker crash publication could not be drained" @@ -404,8 +584,9 @@ test_empty_prefix_mate_preserves_other_mate_receipt() { printf '%s\t9\tcheck\trouted\tcheck: routed row\n' "$epoch" > "$stalled/state/.wake-queue" row_before="$dir/foreign-before" cp "$stalled/state/.wake-queue" "$row_before" + printf '%s\t%s-9\n' "$(( $(date +%s) - 2 ))" "$epoch" > "$state/.secondmate-wake-progress-ios-ui" append_wake "$state" check "secondmate-wake-loop-ios-ui-$epoch-9" \ - "check: secondmate wake-loop stalled: mate=ios-ui row=9 age=10s" \ + "check: secondmate wake-loop stalled: mate=ios-ui row=9 idle=2s" \ || fail "could not seed the ios-ui stall publication" FM_STATE_OVERRIDE="$state" "$DRAIN" > "$dir/drain.out" 2> "$dir/drain.err" \ || fail "ios-ui stall publication could not be drained" @@ -703,6 +884,164 @@ test_main_drain_excludes_rows_already_granted_to_branch() { pass "main drain and acknowledgement exclude an active branch grant" } +# The pending-warning condition and what a drain can actually present must name +# the same rows. A row reserved by a live branch grant is invisible to a main +# drain by design, so counting it as "queued for main" told main to run a drain +# that could only print nothing - no row, no acknowledgement command - on every +# guarded command, for as long as the branch held the grant. +test_main_is_never_told_to_drain_rows_only_the_branch_owns() { + local dir state out err sequence generation + dir=$(make_case main-not-told-to-drain-branch-rows) + state="$dir/state" + printf 'window=test:fm-x\nkind=ship\n' > "$state/x.meta" + + append_wake "$state" stale "fleet:w2:p3" "stale: fleet:w2:p3 (paused, awaiting external)" \ + || fail "stale append failed" + FM_STATE_OVERRIDE="$state" "$GRANT" activate "$$" held-by-branch || fail "branch owner activation failed" + FM_STATE_OVERRIDE="$state" "$GRANT" publish held-by-branch 1 || fail "branch grant publication failed" + + out="$dir/main-drain.out" + err="$dir/main-drain.err" + FM_STATE_OVERRIDE="$state" "$DRAIN" > "$out" 2> "$err" || fail "main drain failed: $(cat "$err")" + ! grep -Fq "$(printf '\tstale\tfleet:w2:p3\t')" "$out" || fail "main drain presented a branch-granted row" + grep -Fq 'WAKE ROWS HELD BY SUPERVISION BRANCH' "$out" \ + || fail "main drain went silent instead of naming who holds the queued rows" + ! grep -Fq 'WAKE_ACK_REQUIRED' "$err" || fail "main drain offered an acknowledgement for a row it never presented" + ! grep -Fq 'queued wakes pending' "$err" \ + || fail "main was told to drain rows only the branch can present" + FM_STATE_OVERRIDE="$state" "$GUARD" 2> "$dir/guard-held.err" || fail "guard failed while the branch held the rows" + ! grep -Fq 'queued wakes pending' "$dir/guard-held.err" \ + || fail "guard counted branch-held rows as pending for main" + grep -Fq 'wake rows held by the live supervision branch' "$dir/guard-held.err" \ + || fail "guard went silent about a non-empty queue instead of naming the branch as its holder" + grep -Fq 'do not drain them from here' "$dir/guard-held.err" \ + || fail "the held advisory did not say the rows must not be drained from here" + grep -Fq "$(printf '\tstale\tfleet:w2:p3\t')" "$state/.wake-queue" \ + || fail "the branch-held row must stay durable for its own owner" + + # Disconfirming half: the same row, same kind, same stopped endpoint, with the + # grant released. Nothing about the row makes it unpresentable - only the + # live grant did - so main now presents it with an executable acknowledgement. + FM_STATE_OVERRIDE="$state" "$GRANT" release held-by-branch || fail "branch grant release failed" + out="$dir/main-drain-after.out" + err="$dir/main-drain-after.err" + FM_STATE_OVERRIDE="$state" "$DRAIN" > "$out" 2> "$err" || fail "main drain failed after release: $(cat "$err")" + grep -Fq "$(printf '\tstale\tfleet:w2:p3\t')" "$out" || fail "main drain omitted the released row" + ! grep -Fq 'WAKE ROWS HELD BY SUPERVISION BRANCH' "$out" \ + || fail "main drain reported a hold that no longer exists" + sequence=$(sed -n 's/^WAKE_ACK_REQUIRED:.*--ack-through \([0-9][0-9]*\) --recovery-generation [A-Za-z0-9._-][A-Za-z0-9._-]*$/\1/p' "$err") + generation=$(sed -n 's/^WAKE_ACK_REQUIRED:.*--ack-through [0-9][0-9]* --recovery-generation \([A-Za-z0-9._-][A-Za-z0-9._-]*\)$/\1/p' "$err") + [ -n "$sequence" ] && [ -n "$generation" ] || fail "the released row was presented without an acknowledgement command" + grep -Fq 'queued wakes pending' "$err" || fail "guard stopped warning about a row main can actually drain" + ! grep -Fq 'wake rows held by the live supervision branch' "$err" \ + || fail "guard kept advising about a hold that was already released" + FM_STATE_OVERRIDE="$state" "$DRAIN" --ack-through "$sequence" --recovery-generation "$generation" \ + || fail "acknowledgement of the released row failed" + [ ! -s "$state/.wake-queue" ] || fail "the acknowledged row stayed queued" + + pass "a branch-held row raises no queued-wake warning for main, and the same row is presented and acknowledged once the grant clears" +} + +# The pending-warning condition must also survive a queue nobody could read: a +# queue that exists but cannot be counted is not evidence that it was drained. +# The per-actor count runs awk over the queue, and awk implementations differ on +# whether a failed input open aborts before the END rule; one that reaches END +# reports a 0 count for a queue that was never proved empty. +test_uncountable_queue_still_raises_the_pending_alarm() { + local dir state awkbin real_awk + dir=$(make_case uncountable-queue) + state="$dir/state" + awkbin="$dir/awkbin" + mkdir -p "$awkbin" + printf 'window=test:fm-x\nkind=ship\n' > "$state/x.meta" + + # An awk that still runs its END rule after failing to open its input: it + # prints a 0 count and exits non-zero. Every other invocation is the real awk. + real_awk=$(command -v awk) || fail "no awk on PATH" + cat > "$awkbin/awk" < "$dir/unreadable.err" \ + || fail "guard failed on an unreadable queue" + grep -Fq 'queued wakes pending' "$dir/unreadable.err" \ + || fail "a queue that could not be counted silenced the queued-wake alarm" + chmod 600 "$state/.wake-queue" || fail "could not restore the queue" + + # Disconfirming half: the same fake awk over a queue that is readable and + # provably empty stays silent, so the warning above came from the failed count + # and not from the fake awk itself. + : > "$state/.wake-queue" + PATH="$awkbin:$PATH" FM_STATE_OVERRIDE="$state" "$GUARD" 2> "$dir/empty.err" \ + || fail "guard failed on an empty queue" + ! grep -Fq 'queued wakes pending' "$dir/empty.err" \ + || fail "a provably empty queue raised the queued-wake alarm" + + pass "a queue that cannot be counted keeps the queued-wake alarm up" +} + +# A row that lost its structure can never be claimed, presented, or named by an +# --ack-through cutoff, while it still counts as queued: without retirement it +# wedges the queue permanently and keeps waking supervision. +test_unconsumable_rows_are_retired_instead_of_wedging_the_queue() { + local dir state out err sequence generation + dir=$(make_case unconsumable-row-retirement) + state="$dir/state" + printf 'window=test:fm-x\nkind=ship\n' > "$state/x.meta" + + append_wake "$state" signal "task-a.status" "signal: task-a" || fail "signal append failed" + printf '1788792074\t574\tstale\tfleet:w2:p3\n' >> "$state/.wake-queue" + printf '1788792075\tnot-a-sequence\tstale\tfleet:w2:p4\tstale: fleet:w2:p4\n' >> "$state/.wake-queue" + + # A branch actor never repairs the queue: it may only touch its own grant. + FM_STATE_OVERRIDE="$state" "$GRANT" activate "$$" retire-scope || fail "branch owner activation failed" + FM_STATE_OVERRIDE="$state" "$GRANT" publish retire-scope 1 || fail "branch grant publication failed" + FM_STATE_OVERRIDE="$state" FM_SUPERVISION_ACTOR=branch "$DRAIN" > "$dir/branch.out" 2> "$dir/branch.err" \ + || fail "branch drain failed: $(cat "$dir/branch.err")" + ! grep -Fq 'retired' "$dir/branch.err" || fail "a branch drain retired rows outside its grant" + [ "$(awk 'END { print NR }' "$state/.wake-queue")" -eq 3 ] \ + || fail "a branch drain changed rows it was never granted" + FM_STATE_OVERRIDE="$state" "$GRANT" release retire-scope || fail "branch grant release failed" + + FM_STATE_OVERRIDE="$state" "$GUARD" 2> "$dir/guard-before.err" || fail "guard failed with unusable rows queued" + grep -Fq 'queued wakes pending' "$dir/guard-before.err" \ + || fail "guard stayed silent about rows main still has to clear" + ! grep -Fq 'wake rows held by the live supervision branch' "$dir/guard-before.err" \ + || fail "guard advised a branch hold for rows no grant covers" + + out="$dir/main.out" + err="$dir/main.err" + FM_STATE_OVERRIDE="$state" "$DRAIN" > "$out" 2> "$err" || fail "main drain failed: $(cat "$err")" + grep -Fq 'retired 2 unusable queue row(s)' "$err" || fail "main drain did not report the rows it retired" + grep -Fq "$(printf '1788792074\t574\tstale\tfleet:w2:p3')" "$err" \ + || fail "the retired row's content was discarded instead of reported" + grep -Fq "$(printf '1788792075\tnot-a-sequence\tstale\tfleet:w2:p4\tstale: fleet:w2:p4')" "$err" \ + || fail "the second retired row's content was discarded instead of reported" + grep -Fq "$(printf '\tsignal\ttask-a.status\t')" "$out" || fail "retirement dropped a usable row" + [ "$(awk 'END { print NR }' "$state/.wake-queue")" -eq 1 ] || fail "unusable rows survived the drain" + sequence=$(sed -n 's/^WAKE_ACK_REQUIRED:.*--ack-through \([0-9][0-9]*\) --recovery-generation [A-Za-z0-9._-][A-Za-z0-9._-]*$/\1/p' "$err") + generation=$(sed -n 's/^WAKE_ACK_REQUIRED:.*--ack-through [0-9][0-9]* --recovery-generation \([A-Za-z0-9._-][A-Za-z0-9._-]*\)$/\1/p' "$err") + [ -n "$sequence" ] && [ -n "$generation" ] || fail "the usable row was presented without an acknowledgement command" + FM_STATE_OVERRIDE="$state" "$DRAIN" --ack-through "$sequence" --recovery-generation "$generation" \ + || fail "acknowledgement failed" + [ ! -s "$state/.wake-queue" ] || fail "the queue stayed wedged after acknowledgement" + FM_STATE_OVERRIDE="$state" "$GUARD" 2> "$dir/guard-after.err" || fail "guard failed after the queue drained" + ! grep -Fq 'queued wakes pending' "$dir/guard-after.err" || fail "guard kept warning about an empty queue" + + pass "structurally unusable rows are retired by main alone, leaving every remaining row presentable and acknowledgeable" +} + test_branch_grant_refuses_rows_already_claimed_by_main() { local dir state rc dir=$(make_case branch-refuses-main-claim) @@ -1573,7 +1912,10 @@ test_subshell_lock_ownership_without_bashpid test_bounded_lock_handoff_after_contention test_live_presentation_holder_is_deadlined_without_weakening_ack test_malformed_presentation_lock_reports_acquire_failure -test_secondmate_foreign_queue_stall_is_one_shot_and_read_only +test_secondmate_foreign_queue_stall_tracks_progress_and_alerts_once +test_secondmate_declared_pause_rows_do_not_feed_stall_escalation +test_secondmate_reprovisioned_queue_starts_a_fresh_interval +test_secondmate_active_turn_defers_stall_until_the_turn_ends test_secondmate_stall_marker_rejects_symlink test_acknowledged_stall_publication_survives_pre_marker_crash test_empty_prefix_mate_preserves_other_mate_receipt @@ -1592,6 +1934,9 @@ test_enrichment_preserves_all_unread_lines_and_status_file_failures test_slow_annotation_does_not_block_append_and_deleted_file_fails_open test_branch_actor_scoped_ack_never_swallows_a_main_owned_row test_main_drain_excludes_rows_already_granted_to_branch +test_main_is_never_told_to_drain_rows_only_the_branch_owns +test_uncountable_queue_still_raises_the_pending_alarm +test_unconsumable_rows_are_retired_instead_of_wedging_the_queue test_branch_grant_refuses_rows_already_claimed_by_main test_actor_filter_precedes_same_key_deduplication test_main_reclaims_a_grant_whose_branch_owner_exited diff --git a/tests/fm-watch-triage.test.sh b/tests/fm-watch-triage.test.sh index c82234b6c68..029b4a491f9 100755 --- a/tests/fm-watch-triage.test.sh +++ b/tests/fm-watch-triage.test.sh @@ -2357,6 +2357,274 @@ test_live_declared_wait_churn_honors_the_resurface_throttle() { pass "a parked live worker surfaces once, absorbs pane churn for the whole re-surface window, then re-surfaces when it elapses" } +# --- work the captain is already holding: pane churn must not re-alarm ------- +# The other record of a legitimate wait. The declared-wait bound above reads the +# status LINE, and a delivered task's line stays `done: PR ...` while the wait +# itself lives in the BACKLOG, written there by bin/fm-captain-hold.sh. No line +# predicate can see that record, so both stale alarms - the captain-relevant one +# and the inconclusive one - re-fired on every new pane hash for as long as the +# captain was deciding, which is the 2026-09 loop observed on delivered work +# awaiting their merge word. +# Pinned here, in both directions: while the call stands the first sight still +# alarms, further sights of the SAME call and status-log state are absorbed, and +# a new pane hash after the window's end alarms once more; and the identical +# fixture WITHOUT the hold keeps alarming on every hash, because a bound that +# swallowed an unheld delivery or blocker would be worse than the churn it removes. +# +# The backlog is real rather than a fixture file: bin/fm-captain-hold.sh is the +# only writer of a hold and tasks-axi the only reader, so a hand-written row +# would pin this test's idea of a hold instead of the one the watcher consults. +# +# Cost: every case below drives churn through ONE watcher process rather than +# relaunching per pane change. Watcher startup dominates a round here, and an +# absorbing watcher stays in its poll loop across churn in production anyway, so +# the cheaper shape is also the more faithful one. + +# The window key every hold fixture uses, derived the way fm-watch.sh derives it. +hold_key() { + printf '%s' test:fm-held-merge | tr ':/.' '___' +} + +# bin/fm-captain-hold.sh against a hold fixture's own home. +run_hold() { # + local dir=$1 + shift + FM_HOME="$dir" FM_STATE_OVERRIDE="$dir/state" FM_DATA_OVERRIDE="$dir/data" \ + FM_CONFIG_OVERRIDE="$dir/config" "$ROOT/bin/fm-captain-hold.sh" "$@" >/dev/null 2>&1 +} + +make_hold_home() { # + local name=$1 line=$2 hold=$3 dir state + dir=$(make_case "$name"); state="$dir/state" + mkdir -p "$dir/data" "$dir/config" + cp "$ROOT/.tasks.toml" "$dir/.tasks.toml" || return 1 + printf '## In flight\n\n## Queued\n\n## Done\n' > "$dir/data/backlog.md" + (cd "$dir" && tasks-axi add held-merge 'delivered work' --file data/backlog.md) >/dev/null 2>&1 \ + || return 1 + if [ "$hold" = hold ]; then + run_hold "$dir" hold held-merge --reason 'awaiting the captain on the merge' || return 1 + fi + printf 'window=test:fm-held-merge\nkind=ship\nharness=grok\nbackend=tmux\n' \ + > "$state/held-merge.meta" + printf '%s\n' "$line" > "$state/held-merge.status" + printf '%s' "$(seen_sig "$state/held-merge.status")" > "$state/.seen-held-merge_status" + printf '%s\n' "$dir" +} + +# Launch one watcher against a hold fixture, armed the way parked_watch_round +# arms one, plus the home the backlog read resolves against. The crew reads +# stopped: a delivered worker's agent has exited, and that is the population +# whose alarm the call must bound. The pid lands in HOLD_WATCH_PID rather than on +# stdout: a command substitution would background the watcher inside a subshell, +# leaving the caller unable to wait on or reap its own watcher. +HOLD_WATCH_PID= +hold_watch_launch() { # + local dir=$1 out=$2 capture=$3 + PATH="$dir/fakebin:$PATH" FM_FAKE_TMUX_WINDOW=test:fm-held-merge \ + FM_FAKE_TMUX_CAPTURE="$capture" FM_FAKE_TMUX_CURRENT_COMMAND=zsh \ + FM_FAKE_CREW_STATE='state: stopped · source: pane · bare shell' \ + FM_WATCH_HANDLING_SUCCESSOR=1 \ + FM_HOME="$dir" FM_DATA_OVERRIDE="$dir/data" FM_CONFIG_OVERRIDE="$dir/config" \ + FM_STATE_OVERRIDE="$dir/state" FM_CREW_STATE_BIN="$dir/fakebin/fm-crew-state.sh" \ + FM_PAUSE_RESURFACE_SECS="${FM_HOLD_PAUSE_RESURFACE_SECS:-999}" FM_POLL=1 FM_SIGNAL_GRACE=1 \ + FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" >> "$out" 2>&1 & + HOLD_WATCH_PID=$! +} + +# One sighting that must surface and exit the cycle. +hold_watch_surface() { # + local dir=$1 out=$2 capture=$3 text=$4 + printf '%s\n' "$text" > "$capture" + hold_watch_launch "$dir" "$out" "$capture" + wait_for_exit "$HOLD_WATCH_PID" 100 || { reap "$HOLD_WATCH_PID"; return 1; } + return 0 +} + +# successive pane changes driven through ONE watcher, each given three +# poll cycles: one to see the new hash, one to count it stable and classify, one +# to prove the classification held. The watcher must stay in the loop throughout. +hold_watch_churn() { #