diff --git a/bin/fm-session-start.sh b/bin/fm-session-start.sh index 37b4909161f..2d5e663b56d 100755 --- a/bin/fm-session-start.sh +++ b/bin/fm-session-start.sh @@ -47,8 +47,13 @@ # every state/*.meta, a bounded state/*.status tail, # the away posture (state/.afk-contract and the legacy # state/.afk daemon flag), and a cheap per-task -# endpoint-liveness read: -# read-only, always runs. +# endpoint-liveness read, each bounded and crash- +# isolated so one task's read can never abort the +# digest: read-only, always runs. The per-task reads +# run serially, so with a wedged backend the stage's +# ceiling is tasks x the per-read bound +# (FM_SESSION_START_ENDPOINT_TIMEOUT, default 10s) and +# can itself reach the digest's runtime bound. # 7. network checks - the result of the deferred network stage started back at # step 1, harvested WITHOUT waiting for it. # 8. context digest - data/projects.md, data/secondmates.md, data/captain.md, @@ -59,7 +64,9 @@ # block and deliberately never arms the watcher itself. # # Those nine names are also the runtime-bound stage list below, so a truncated -# startup can name exactly which of them never ran. +# startup can name exactly which of them never ran - and the parent banners +# EVERY nonzero child exit, not only the bound: a child that dies or is killed +# mid-stage must never truncate the digest silently. # # NO NETWORK ON THE BLOCKING PATH. This digest runs on a session-open hook that # blocks session initialization, so anything it waits for is time the captain @@ -169,17 +176,20 @@ # session initialization or Pi's first provider preflight while it runs, so an # unbounded digest is no longer merely slow - it can strand a whole session or # first turn behind one hung subprocess. Every remaining step is local, but -# local is not the same as bounded: tool version probes, the backlog listing, -# and the per-task endpoint reads are all unbounded subprocesses. So the whole -# digest still runs as ONE bounded child of this script -# (FM_SESSION_START_TIMEOUT, default 120s). The deferred network stage +# local is not the same as bounded: tool version probes and the backlog +# listing are unbounded subprocesses, while each per-task endpoint read runs +# in its own crash-isolated child under FM_SESSION_START_ENDPOINT_TIMEOUT +# (default 10s). So the whole digest still runs as ONE bounded child of this +# script (FM_SESSION_START_TIMEOUT, default 120s). The deferred network stage # deliberately sits OUTSIDE that bound, # in its own process group under its own aggregate deadline, so a truncated # digest neither waits for it nor orphans it unbounded. The # child writes the digest straight to this script's stdout, so everything it -# emitted before the bound was hit is already delivered; the parent then prints -# a loud STARTUP TRUNCATED banner naming the stage that did not finish and the -# sections that were therefore never emitted, and still exits 0. The child +# emitted before the child stopped is already delivered; the parent then prints +# a loud STARTUP TRUNCATED banner on ANY nonzero child exit - the runtime bound +# or an unexpected child death, named with its exit status - naming the stage +# that did not finish and the sections that were therefore never emitted, and +# still exits 0. The child # records its progress in FM_SESSION_START_STAGE_FILE, which is also the flag # that tells a child it is the child - the parent never recurses. # Hosts without timeout, gtimeout, or perl use the shared pure-Bash watchdog, so @@ -279,7 +289,8 @@ if [ -z "${FM_SESSION_START_STAGE_FILE:-}" ]; then # A non-positive or non-numeric budget is not a budget (`timeout 0` disables # the deadline outright), so an unusable value falls back to the default # rather than silently removing the bound. - case "$SESSION_START_BUDGET" in ''|*[!0-9]*|0) SESSION_START_BUDGET=120 ;; esac + case "$SESSION_START_BUDGET" in ''|*[!0-9]*) SESSION_START_BUDGET=120 ;; esac + [ "$SESSION_START_BUDGET" -gt 0 ] 2>/dev/null || SESSION_START_BUDGET=120 SESSION_START_STAGE_FILE=$(mktemp "${TMPDIR:-/tmp}/fm-session-start-stage.XXXXXX" 2>/dev/null) || SESSION_START_STAGE_FILE= if [ -z "$SESSION_START_STAGE_FILE" ]; then # Without a breadcrumb the bound still holds; only the banner's precision @@ -306,7 +317,11 @@ if [ -z "${FM_SESSION_START_STAGE_FILE:-}" ]; then "$SCRIPT_DIR/fm-session-start.sh" fi SESSION_START_RC=$? - if [ "$SESSION_START_RC" -eq 124 ]; then + # ANY nonzero child exit is a truncation: the banner contract promises that + # a stage that cannot print is named. Exit 124 is the bound firing; any + # other status means the child died or was killed mid-stage, which truncates + # silently when unbanned - the parent must banner it, never exit 0 around it. + if [ "$SESSION_START_RC" -ne 0 ]; then SESSION_START_LAST_STAGE=$(cat "$SESSION_START_STAGE_FILE" 2>/dev/null) || SESSION_START_LAST_STAGE= [ -n "$SESSION_START_LAST_STAGE" ] || SESSION_START_LAST_STAGE=unknown SESSION_START_PENDING=$( @@ -316,14 +331,23 @@ if [ -z "${FM_SESSION_START_STAGE_FILE:-}" ]; then [ -n "${SESSION_START_PENDING# }" ] || SESSION_START_PENDING='(unknown - the digest may be incomplete anywhere)' BAR='●━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━' printf '\n%s\n' "$BAR" - printf '● STARTUP TRUNCATED - SESSION START HIT ITS %ss RUNTIME BOUND\n' "$SESSION_START_BUDGET" + if [ "$SESSION_START_RC" -eq 124 ]; then + printf '● STARTUP TRUNCATED - SESSION START HIT ITS %ss RUNTIME BOUND\n' "$SESSION_START_BUDGET" + else + printf '● STARTUP TRUNCATED - SESSION START DIED UNEXPECTEDLY (exit %s, not its runtime bound)\n' "$SESSION_START_RC" + fi printf '● It stopped during the "%s" stage, so everything above is COMPLETE\n' "$SESSION_START_LAST_STAGE" printf '● only up to that point.\n' printf '● RECONCILE these stages before acting on anything they would have shown:\n' printf '● %s\n' "${SESSION_START_PENDING% }" printf '● Rerun bin/fm-session-start.sh now to finish taking the helm. If it truncates\n' - printf '● again, raise FM_SESSION_START_TIMEOUT and report the slow stage - a stage that\n' - printf '● cannot finish inside the bound is a fleet problem, not a reporting detail.\n' + if [ "$SESSION_START_RC" -eq 124 ]; then + printf '● again, raise FM_SESSION_START_TIMEOUT and report the slow stage - a stage that\n' + printf '● cannot finish inside the bound is a fleet problem, not a reporting detail.\n' + else + printf '● again, report the exit status and the stage - raising the runtime bound\n' + printf '● cannot help a digest that died, and a stage that dies is a fleet problem.\n' + fi printf '%s\n' "$BAR" fi rm -f "$SESSION_START_STAGE_FILE" 2>/dev/null || true @@ -358,6 +382,12 @@ STATUS_TAIL=${FM_SESSION_START_STATUS_TAIL:-5} case "$STATUS_TAIL" in ''|*[!0-9]*) STATUS_TAIL=5 ;; esac QUEUED_LIMIT=${FM_SESSION_START_QUEUED_LIMIT:-20} case "$QUEUED_LIMIT" in ''|*[!0-9]*|0) QUEUED_LIMIT=20 ;; esac +# One per-task endpoint read may never outlive this bound: a hung backend CLI +# becomes that task's endpoint: error line instead of the digest's whole +# runtime budget. +ENDPOINT_TIMEOUT=${FM_SESSION_START_ENDPOINT_TIMEOUT:-10} +case "$ENDPOINT_TIMEOUT" in ''|*[!0-9]*) ENDPOINT_TIMEOUT=10 ;; esac +[ "$ENDPOINT_TIMEOUT" -gt 0 ] 2>/dev/null || ENDPOINT_TIMEOUT=10 BACKLOG_FIELDS=blocked_by,hold_kind,hold_reason RULE='================================================================================' @@ -537,6 +567,23 @@ print_status_tail() { done < <(tail -n "$STATUS_TAIL" "$status") } +# fm_session_start_endpoint_read [expected-label]: ONE +# bounded, crash-isolated endpoint-liveness read. The read runs in its own +# bash under fm_run_timed's bound instead of in this digest process, because +# a per-task backend liveness read that dies mid-read would otherwise take +# every later stage with it. Isolation turns any death, hang, or nonzero +# surprise in one task's read into that task's own endpoint line - never a +# silently missing rest of digest. The inner bash re-sources fm-backend.sh +# per read; that cost is a few milliseconds per task and buys the isolation. +fm_session_start_endpoint_read() { # [expected-label] + local backend=$1 target=$2 label=${3:-} + # shellcheck disable=SC2016 # Positional parameters expand inside the child bash, not here. + fm_run_timed "$ENDPOINT_TIMEOUT" bash -c ' + . "$1" + fm_backend_target_exists "$2" "$3" "$4" + ' _ "$SCRIPT_DIR/fm-backend.sh" "$backend" "$target" "$label" +} + hash_file_sha256() { local file=$1 digest [ -f "$file" ] || return 1 @@ -845,8 +892,16 @@ for meta in "$STATE"/*.meta; do target=$(fm_backend_target_of_meta "$meta") if [ -n "$window" ]; then backend=$(fm_backend_of_meta "$meta") - if fm_backend_target_exists "$backend" "${target:-$window}" "fm-$id"; then + endpoint_rc=0 + fm_session_start_endpoint_read "$backend" "${target:-$window}" "fm-$id" || endpoint_rc=$? + # Only the timeout owner's own statuses mean the read itself failed: 124 is + # the bound firing and >=128 is a signal death. Every other nonzero status + # is the probe's own verdict that the endpoint is gone. + if [ "$endpoint_rc" -eq 0 ]; then printf 'endpoint: alive (backend=%s window=%s)\n' "$backend" "$window" + elif [ "$endpoint_rc" -eq 124 ] || [ "$endpoint_rc" -ge 128 ]; then + printf 'endpoint: error (backend=%s window=%s - the endpoint read died or hit its %ss bound; the digest continued past it)\n' \ + "$backend" "$window" "$ENDPOINT_TIMEOUT" else printf 'endpoint: dead (backend=%s window=%s)\n' "$backend" "$window" fi diff --git a/bin/fm-timeout-lib.sh b/bin/fm-timeout-lib.sh index 4e47aaa9ab6..a785ad8b793 100644 --- a/bin/fm-timeout-lib.sh +++ b/bin/fm-timeout-lib.sh @@ -13,9 +13,15 @@ # fm_run_timed [args...] # Runs the command with a hard bound. Exit status is the command's own, # except 124, which means the bound was hit (GNU timeout's convention, -# reproduced by the perl and bash fallbacks). A signal-death status the -# wrapper records while the runner already reports the bound is the -# bound's own TERM, not the command's exit, and is reported as 124 too. +# reproduced by the perl and bash fallbacks), and a command killed by +# signal n, which reports 128+n on every mechanism - so a SIGKILLed child +# is 137 and a SIGTERMed one 143, never the 0 a caller would read as +# success. A signal-death status the wrapper records while the runner +# already reports the bound is the bound's own TERM, not the command's +# exit, and is reported as 124 too. Only 137 raised by GNU/BSD timeout's +# own KILL escalation, with no status recorded by the bounded command, +# also collapses into 124: there it means the bound fired, not that the +# command chose to die. # # fm_exec_timed [args...] # Replaces the calling shell with the bounded command, so it must be the @@ -179,7 +185,7 @@ fm_run_timed() { # timeout) fm_run_external_timeout timeout "$seconds" "$@" ;; gtimeout) fm_run_external_timeout gtimeout "$seconds" "$@" ;; perl) - perl -e 'my $t = shift; my $pid = fork; die "fork failed" unless defined $pid; if (!$pid) { setpgrp(0, 0); exec @ARGV } local $SIG{ALRM} = sub { kill "TERM", -$pid; select undef, undef, undef, 0.2; kill "KILL", -$pid; exit 124 }; alarm $t; waitpid $pid, 0; exit($? >> 8)' \ + perl -e 'my $t = shift; my $pid = fork; die "fork failed" unless defined $pid; if (!$pid) { setpgrp(0, 0); exec @ARGV } local $SIG{ALRM} = sub { kill "TERM", -$pid; select undef, undef, undef, 0.2; kill "KILL", -$pid; exit 124 }; alarm $t; waitpid $pid, 0; exit(($? & 127) ? 128 + ($? & 127) : $? >> 8)' \ "$seconds" "$@" ;; bash) fm_run_bash_timeout "$seconds" "$@" ;; diff --git a/docs/configuration.md b/docs/configuration.md index 5e3bf53bc5c..9dfa54fb80c 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -2247,6 +2247,7 @@ FM_ZELLIJ_SESSION=firstmate # zellij-only: named session for normal backend ops CMUX_SOCKET_PASSWORD= # cmux-only: socket password fallback when config/cmux-socket-password is absent (docs/cmux-backend.md) FM_SESSION_START_STATUS_TAIL=5 # state/*.status lines printed per task in the session-start digest; each line is capped by bin/fm-line-cap-lib.sh FM_SESSION_START_QUEUED_LIMIT=20 # plain queued backlog rows in the session-start digest; in-flight, held, and blocked rows are never bounded and done rows are never listed +FM_SESSION_START_ENDPOINT_TIMEOUT=10 # seconds bounding each per-task endpoint liveness read in the session-start digest (bin/fm-session-start.sh); nonpositive or invalid values fall back to 10; a read that hits the bound or dies becomes that task's own `endpoint: error` line and the digest continues FM_BACKLOG_ROW_TIMEOUT_SECS=10 # seconds bounding each backlog row read (bin/fm-backlog-transition-lib.sh); nonpositive or invalid values fall back to 10; the first bound hit latches the sweep so later reads return immediately, each still naming its own item FM_BOOTSTRAP_DETECT_ONLY=0 # internal/read-only session-start mode: skip bootstrap's mutating sweeps and print advisory TANGLE wording FM_BOOTSTRAP_NETWORK=all # internal session-start phase split: all, skip (local steps only), or only (network steps only); see bin/fm-bootstrap.sh diff --git a/docs/sessionstart-nudge.md b/docs/sessionstart-nudge.md index b2cfe3984cc..bf446312524 100644 --- a/docs/sessionstart-nudge.md +++ b/docs/sessionstart-nudge.md @@ -141,10 +141,13 @@ Some digest work remains local but unbounded: - Tool version probes. - The backlog listing. -- The per-task endpoint reads. So the whole digest still runs as one bounded child, default 120s via `FM_SESSION_START_TIMEOUT`. +Each per-task endpoint liveness read runs serially in its own crash-isolated child, bounded by `FM_SESSION_START_ENDPOINT_TIMEOUT` (default 10s; a non-numeric or zero value falls back to the default). +So a read that hangs or dies becomes that task's own `endpoint: error` line and the digest continues. +With a wedged backend the stage's ceiling is tasks times that per-read bound and can itself reach the digest bound. + The per-item backlog row reads inside bootstrap's reconcile and close-replay sweeps are the exception. Each of those reads is bounded by `FM_BACKLOG_ROW_TIMEOUT_SECS` (default 10s) through `bin/fm-backlog-transition-lib.sh`. The first bound hit latches the sweep. @@ -153,16 +156,18 @@ Later reads in that sweep then return immediately while still naming their own i When timeout, gtimeout, and perl are unavailable, the shared timeout owner falls back to a pure-Bash process-group watchdog. So no supported host runs the digest unbounded. -### When the bound is hit +### When the child stops early The child streams into the native transport as it runs. -So everything emitted before the bound was hit is retained for delivery. -The parent then prints a `STARTUP TRUNCATED` banner that names: +So everything emitted before the child stopped is retained for delivery. +The parent then prints a `STARTUP TRUNCATED` banner on any nonzero child exit, not only the bound, that names: - The stage that did not finish. - The stages that were therefore never emitted. +- Whether the child hit its bound or died unexpectedly with its exit status. The parent still exits 0. +The regression evidence for both shapes is in [`docs/verification/supervision.md`](verification/supervision.md#per-task-endpoint-reads-cannot-truncate-the-digest). The registered hook timeouts sit above that budget, so the harness never preempts the banner. The deferred startup stage deliberately runs in its own process group under its own deadline. diff --git a/docs/verification/supervision.md b/docs/verification/supervision.md index 3e3d0e0d607..9fe29503ae8 100644 --- a/docs/verification/supervision.md +++ b/docs/verification/supervision.md @@ -203,6 +203,23 @@ The Ahoy first-message boundary was reverified on 2026-07-22 with Pi 0.81.1 and Marked current operational input and the two exact legacy compatibility shapes selected Bearings, while genuine near-miss captain messages remained real boundaries. The detailed reconciliation and task chronology stay in the private audit report and PR evidence. +### Per-task endpoint reads cannot truncate the digest + +A per-task backend endpoint liveness read that dies mid-read inside the digest process takes every later stage with it, and a parent wrapper that banners only the runtime-bound exit stays silent about the missing sections. +The digest now runs each per-task endpoint read in its own bounded child (`FM_SESSION_START_ENDPOINT_TIMEOUT`, default 10s) whose death, hang, or nonzero surprise becomes that task's own `endpoint: error` line, and the parent wrapper banners ANY nonzero child exit, naming the stage and the abnormal exit status. +Verified on 2026-09-27 with the deterministic process-tree tests that reproduce both failure shapes with real processes and no harness: + +```sh +tests/fm-session-start.test.sh +# ok - a killed per-task endpoint read becomes that task's error line and the digest completes +# ok - a hung per-task endpoint read hits its configured bound, reports the task, and leaves nothing stuck +# ok - a digest child killed mid-stage is bannered by the parent, which still exits 0 +``` + +The kill test's fake `ps` walks real `/proc` ancestry to TERM the digest bash itself mid-lock-stage, so the parent-wrapper banner path is exercised end to end rather than asserted from output shape alone. +Both process-tree cases therefore need a readable `/proc` and print a skip line without it, and the companion case that pins a signal death to a nonzero status on the perl timeout mechanism skips when `perl` is absent. +These guarantees are process semantics, not vendor-emitted signals, so no live-harness guard is owed; the same suite is the refresh command. + ## Semantic busy state The per-adapter semantic sources behind [`bin/fm-busy-lib.sh`](../../bin/fm-busy-lib.sh) were live-verified on 2026-07-28 against firstmate-launched workers wired exactly as `fm-spawn` writes them. diff --git a/tests/fm-session-start.test.sh b/tests/fm-session-start.test.sh index 490fec1bace..1863338008e 100755 --- a/tests/fm-session-start.test.sh +++ b/tests/fm-session-start.test.sh @@ -488,16 +488,64 @@ SH chmod +x "$fakebin/herdr" } -# make_fake_herdr : `herdr pane get ` succeeds only -# for the given pane id - the exact primitive fm_backend_target_exists uses -# for a herdr endpoint liveness read. No version/server-start calls: a +# make_fake_herdr [odd-status-pane]: `herdr pane get +# ` succeeds only for the given pane id - the exact primitive +# fm_backend_target_exists uses for a herdr endpoint liveness read. An +# optional second pane id answers with exit 4, the shape backend probes +# produce for a gone surface without normalising to 1 (jq -e on empty input, +# orca's ok:false, a missing tmux binary). No version/server-start calls: a # liveness check must never auto-start a server (fm-backend.sh's contract). make_fake_herdr() { - local fakebin=$1 live=$2 + local fakebin=$1 live=$2 odd=${3:-} cat > "$fakebin/herdr" < : like +# make_fake_herdr, but `pane get ` KILLs the shell running the +# endpoint read. The read's shell is the fake's grandparent (fm_backend_herdr_cli's +# stderr-capture subshell sits in between), so the fake walks one /proc hop +# above $PPID. This is the digest-death shape: a per-task herdr liveness +# read whose process died mid-read, which used to take the whole +# session-start digest with it. +make_fake_herdr_deadly_read() { + local fakebin=$1 live=$2 killpane=$3 + cat > "$fakebin/herdr" </dev/null | awk '{print \$2}') + kill -KILL "\$read_shell" 2>/dev/null + exit 0 + fi + [ "\${3:-}" = "$live" ] && exit 0 + exit 1 +fi +exit 1 +SH + chmod +x "$fakebin/herdr" +} + +# make_fake_herdr_hanging_read : like +# make_fake_herdr, but `pane get ` never returns - the backend-CLI +# hang shape the per-task read bound must turn into an error line. +make_fake_herdr_hanging_read() { + local fakebin=$1 live=$2 hangpane=$3 + cat > "$fakebin/herdr" < "$home/state/task-live.meta" printf 'window=sess:p-dead\nkind=ship\nbackend=herdr\n' > "$home/state/task-dead.meta" + printf 'window=sess:p-odd\nkind=ship\nbackend=herdr\n' > "$home/state/task-odd.meta" out=$(run_session_start "$home" "$root" "$fakebin:$BASE_PATH") assert_contains "$out" "endpoint: alive (backend=herdr window=sess:p-live)" "live herdr endpoint not reported alive" assert_contains "$out" "endpoint: dead (backend=herdr window=sess:p-dead)" "dead herdr endpoint not reported dead" + assert_contains "$out" "endpoint: dead (backend=herdr window=sess:p-odd)" \ + "a probe exiting 4 for a gone surface was not reported dead" + assert_not_contains "$out" "endpoint: error (backend=herdr window=sess:p-odd" \ + "a probe exiting 4 was mislabelled as a failed read" + + pass "herdr endpoint liveness is reported per task: alive, dead for exit 1, dead for any other probe status" +} + +test_endpoint_read_death_is_isolated_and_reported() { + local rec root home fakebin out status=0 + [ -r /proc/self/stat ] || { echo "skip: /proc not readable (the read-death shape needs process ancestry)"; return 0; } + rec=$(new_world endpoint-death) + IFS='|' read -r root home fakebin < "$home/state/task-a-doom.meta" + printf 'working: doomed task marker\n' > "$home/state/task-a-doom.status" + printf 'window=sess:p-live\nkind=ship\nbackend=herdr\n' > "$home/state/task-z-live.meta" + + out=$(FM_SESSION_START_ENDPOINT_TIMEOUT=bogus run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? + + expect_code 0 "$status" "one killed endpoint read must not fail the digest" + assert_contains "$out" \ + "endpoint: error (backend=herdr window=sess:p-doom - the endpoint read died or hit its 10s bound; the digest continued past it)" \ + "a killed endpoint read was not reported as that task's own error line" + assert_contains "$out" "endpoint: alive (backend=herdr window=sess:p-live)" \ + "the digest did not continue past the killed read to the next task" + assert_contains "$out" "working: doomed task marker" \ + "the doomed task's status tail was lost along with its endpoint read" + assert_contains "$out" "$(printf '\nCONTEXT\n')" \ + "a killed endpoint read cost the digest its context section" + assert_contains "$out" "NEXT STEP" \ + "a killed endpoint read cost the digest its closing reminder" + assert_not_contains "$out" "STARTUP TRUNCATED - SESSION START" \ + "an isolated endpoint-read death raised the truncation banner" + assert_present "$home/state/.session-start-complete" \ + "a digest that survived a killed endpoint read did not record completion" + + pass "a killed per-task endpoint read becomes that task's error line and the digest completes" +} + +test_endpoint_read_hang_is_bounded_and_reported() { + local rec root home fakebin out status=0 stray + rec=$(new_world endpoint-hang) + IFS='|' read -r root home fakebin < "$home/state/task-a-slow.meta" + printf 'window=sess:p-live\nkind=ship\nbackend=herdr\n' > "$home/state/task-z-live.meta" + + out=$(FM_SESSION_START_ENDPOINT_TIMEOUT=2 run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? + + expect_code 0 "$status" "a hung endpoint read must not fail the digest" + assert_contains "$out" \ + "endpoint: error (backend=herdr window=sess:p-slow - the endpoint read died or hit its 2s bound; the digest continued past it)" \ + "a hung endpoint read was not bounded into that task's own configured bound" + assert_contains "$out" "endpoint: alive (backend=herdr window=sess:p-live)" \ + "the digest did not continue past the hung read to the next task" + assert_contains "$out" "$(printf '\nCONTEXT\n')" \ + "a hung endpoint read cost the digest its context section" + assert_not_contains "$out" "STARTUP TRUNCATED - SESSION START" \ + "a bounded endpoint-read hang raised the whole-digest truncation banner" + + stray=$(pgrep -f "$fakebin/herdr" 2>/dev/null | wc -l | tr -d ' ') + [ "$stray" -eq 0 ] || fail "the per-task read bound left $stray hung herdr process(es) behind" + + pass "a hung per-task endpoint read hits its configured bound, reports the task, and leaves nothing stuck" +} + +test_endpoint_bound_rejects_padded_zero() { + local rec root home fakebin out status=0 stray + rec=$(new_world endpoint-padded-zero) + IFS='|' read -r root home fakebin < "$home/state/task-a-slow.meta" + printf 'window=sess:p-live\nkind=ship\nbackend=herdr\n' > "$home/state/task-z-live.meta" + + out=$(FM_SESSION_START_ENDPOINT_TIMEOUT=00 run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? + + expect_code 0 "$status" "a padded-zero per-read bound must not fail the digest" + assert_contains "$out" \ + "endpoint: error (backend=herdr window=sess:p-slow - the endpoint read died or hit its 10s bound; the digest continued past it)" \ + "a padded-zero bound did not fall back to the 10s default, so the hung read went unbounded" + assert_contains "$out" "$(printf '\nCONTEXT\n')" \ + "a padded-zero bound cost the digest its context section" + + stray=$(pgrep -f "$fakebin/herdr" 2>/dev/null | wc -l | tr -d ' ') + [ "$stray" -eq 0 ] || fail "the fallback bound left $stray hung herdr process(es) behind" + + pass "a padded-zero per-read bound falls back to the 10s default instead of removing the bound" +} + +test_perl_timeout_fallback_reports_signal_death_nonzero() { + local toolbin cmd rc=0 + command -v perl >/dev/null 2>&1 || { echo "skip: perl not found (this case pins the perl mechanism only)"; return 0; } + toolbin=$(mktemp -d "${TMPDIR:-/tmp}/fm-perl-timeout.XXXXXX") + for cmd in bash perl sleep kill cat rm mktemp; do + command -v "$cmd" >/dev/null 2>&1 && ln -s "$(command -v "$cmd")" "$toolbin/$cmd" + done + PATH="$toolbin" bash -c ' + . "$1/bin/fm-timeout-lib.sh" + [ "$(fm_timeout_mechanism)" = perl ] || { echo "mechanism: $(fm_timeout_mechanism)" >&2; exit 99; } + fm_run_timed 5 bash -c "kill -KILL \$\$" + ' _ "$ROOT" || rc=$? + rm -rf "$toolbin" + expect_code 137 "$rc" "the perl timeout fallback did not report a SIGKILLed child as 128+9" + + pass "the perl timeout fallback reports a signal death as a nonzero status" +} + +test_abnormal_digest_death_banners_and_exits_zero() { + local rec root home fakebin out status=0 + [ -r /proc/self/stat ] || { echo "skip: /proc not readable (the digest-death shape needs process ancestry)"; return 0; } + rec=$(new_world digest-death-banner) + IFS='|' read -r root home fakebin < "$fakebin/ps" </dev/null)" in + *fm-lock.sh*) + pid=\$PPID + target= + matched=0 + for _ in 1 2 3 4 5 6 7 8 9 10 11 12; do + [ -n "\$pid" ] && [ "\$pid" != 1 ] || break + if tr '\\0' '\\n' < /proc/\$pid/environ 2>/dev/null | grep -q '^FM_SESSION_START_STAGE_FILE=' \ + && case "\$(tr '\\0' ' ' < /proc/\$pid/cmdline 2>/dev/null)" in *fm-session-start.sh*) true ;; *) false ;; esac; then + target=\$pid + matched=1 + elif [ "\$matched" -eq 1 ]; then + break + fi + pid=\$(sed 's/^[^)]*) //' /proc/\$pid/stat 2>/dev/null | awk '{print \$2}') + done + [ -n "\$target" ] && kill -TERM "\$target" 2>/dev/null + ;; +esac +exec "$fakebin/ps.real" "\$@" +SH + chmod +x "$fakebin/ps" + + out=$(run_session_start "$home" "$root" "$fakebin:$BASE_PATH") || status=$? + + expect_code 0 "$status" "a digest child that died mid-stage must still let the session open (parent exits 0)" + assert_contains "$out" \ + "STARTUP TRUNCATED - SESSION START DIED UNEXPECTEDLY (exit 143, not its runtime bound)" \ + "a digest child killed mid-stage did not name its abnormal death" + assert_contains "$out" 'stopped during the "lock" stage' \ + "the abnormal-death banner did not name the stage that never finished" + assert_contains "$out" \ + "wake-queue supervision-instructions read-once fleet-state network-checks context next-step" \ + "the abnormal-death banner did not list every stage that never ran" + assert_not_contains "$out" "RUNTIME BOUND" \ + "an abnormal death was misreported as the runtime bound firing" + assert_contains "$out" "report the exit status and the stage" \ + "the abnormal-death banner did not tell the reader to report the exit status" + assert_not_contains "$out" "raise FM_SESSION_START_TIMEOUT" \ + "the abnormal-death banner advised raising a bound that did not fire" + assert_not_contains "$out" "NEXT STEP" \ + "a digest that died mid-stage claimed to have reached its closing reminder" + assert_absent "$home/state/.session-start-complete" \ + "a digest that died mid-stage recorded itself as complete" - pass "herdr endpoint liveness is reported per task: alive for a live pane, dead for a gone one" + pass "a digest child killed mid-stage is bannered by the parent, which still exits 0" } # --- composition: real scripts run, not reimplemented ------------------------ @@ -2727,6 +2972,11 @@ test_status_tail_line_cap test_orphan_status_logs_are_printed test_endpoint_liveness_tmux test_endpoint_liveness_herdr +test_endpoint_read_death_is_isolated_and_reported +test_endpoint_read_hang_is_bounded_and_reported +test_endpoint_bound_rejects_padded_zero +test_perl_timeout_fallback_reports_signal_death_nonzero +test_abnormal_digest_death_banners_and_exits_zero test_composition_invokes_real_scripts test_branch_outcome_replay_respects_captain_barrier_and_lease_sweep test_non_pi_session_start_leaves_branch_state_untouched